diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7677e9bb7..020ef6eee 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -22,6 +22,8 @@ ### Changed +- Split the `AgentSession` implementation into focused session-domain controllers while preserving its public API and runtime behavior. + - Subagents now inherit `async.enabled` and `bash.autoBackground.enabled` from the parent instead of having both force-disabled. Subagent runs complete only after their own background jobs settle and the agent submits a `yield` that postdates every delivered result: a terminal yield with jobs still pending parks the run (recoverable turn stop) instead of completing it, async results are folded in as follow-up turns (with a one-time notice offering `hub` wait/cancel), a result delivered after a yield supersedes that yield and re-runs the yield reminder ladder, and a run that never refreshes a superseded yield fails with the stale payload preserved as salvage. Teardown cancels and awaits surviving jobs before isolation worktree capture and cleanup. - Added ordered `bash.patterns` command approval rules so selected bash commands can be allowed, prompted, or denied by command pattern. - Cache full-session retention transcript incrementally instead of re-formatting the entire message history on every retain cycle ([#4246](https://github.com/can1357/oh-my-pi/issues/4246)) diff --git a/packages/coding-agent/src/modes/utils/context-usage.ts b/packages/coding-agent/src/modes/utils/context-usage.ts index f53edbe15..2ea092585 100644 --- a/packages/coding-agent/src/modes/utils/context-usage.ts +++ b/packages/coding-agent/src/modes/utils/context-usage.ts @@ -41,7 +41,18 @@ export interface ContextBreakdown { snapcompact?: SnapcompactSavingsEstimate; } -const EMPTY_STRING_PARTS: readonly string[] = []; +/** Stable inputs used to cache non-message token estimates. */ +export interface NonMessageTokenSource { + readonly systemPrompt?: string[]; + readonly agent?: { + readonly state?: { + readonly tools?: ReadonlyArray>; + }; + }; + readonly skills?: readonly Skill[]; +} + +const EMPTY_STRING_PARTS: string[] = []; const EMPTY_TOOLS: ReadonlyArray> = []; const EMPTY_SKILLS: readonly Skill[] = []; @@ -111,13 +122,18 @@ interface NonMessageTokenCache { | undefined; } -const nonMessageTokenCache = new WeakMap(); +const NON_MESSAGE_TOKEN_CACHE = Symbol("non-message-token-cache"); -function nonMessageTokenCacheEntry(session: AgentSession): NonMessageTokenCache { +interface CachedNonMessageTokenSource extends NonMessageTokenSource { + [NON_MESSAGE_TOKEN_CACHE]?: NonMessageTokenCache; +} + +function nonMessageTokenCacheEntry(session: NonMessageTokenSource): NonMessageTokenCache { + const cachedSession: CachedNonMessageTokenSource = session; const systemPromptRef = session.systemPrompt ?? EMPTY_STRING_PARTS; const toolsRef = session.agent?.state?.tools ?? EMPTY_TOOLS; const skillsRef = session.skills ?? EMPTY_SKILLS; - let entry = nonMessageTokenCache.get(session); + let entry = cachedSession[NON_MESSAGE_TOKEN_CACHE]; if ( entry && entry.systemPromptRef === systemPromptRef && @@ -127,11 +143,11 @@ function nonMessageTokenCacheEntry(session: AgentSession): NonMessageTokenCache return entry; } entry = { systemPromptRef, toolsRef, skillsRef, tokens: undefined, breakdown: undefined }; - nonMessageTokenCache.set(session, entry); + cachedSession[NON_MESSAGE_TOKEN_CACHE] = entry; return entry; } -export function computeNonMessageTokens(session: AgentSession): number { +export function computeNonMessageTokens(session: NonMessageTokenSource): number { const entry = nonMessageTokenCacheEntry(session); if (entry.tokens !== undefined) return entry.tokens; const systemPromptParts = session.systemPrompt ?? EMPTY_STRING_PARTS; @@ -147,7 +163,7 @@ export function computeNonMessageTokens(session: AgentSession): number { * the status-line fast path intentionally uses the equivalent collapsed total * in `computeNonMessageTokens`. */ -export function computeNonMessageBreakdown(session: AgentSession): { +export function computeNonMessageBreakdown(session: NonMessageTokenSource): { skillsTokens: number; toolsTokens: number; systemContextTokens: number; diff --git a/packages/coding-agent/src/plan-mode/plan-files.ts b/packages/coding-agent/src/plan-mode/plan-files.ts new file mode 100644 index 000000000..838478795 --- /dev/null +++ b/packages/coding-agent/src/plan-mode/plan-files.ts @@ -0,0 +1,40 @@ +import * as fs from "node:fs"; +import * as path from "node:path"; +import { isEnoent } from "@oh-my-pi/pi-utils"; +import { type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls"; +import { normalizeLocalScheme, resolveToCwd } from "../tools/path-utils"; + +/** Reads a plan from a local URL or cwd-relative filesystem path. */ +export async function readPlanFile( + planFilePath: string, + options: { localProtocolOptions: LocalProtocolOptions; cwd: string }, +): Promise { + const resolvedPath = planFilePath.startsWith("local:") + ? resolveLocalUrlToPath(normalizeLocalScheme(planFilePath), options.localProtocolOptions) + : resolveToCwd(planFilePath, options.cwd); + try { + return await Bun.file(resolvedPath).text(); + } catch (error) { + if (isEnoent(error)) return null; + throw error; + } +} + +/** Lists session-local plan files from newest to oldest. */ +export async function listPlanFiles(options: { localProtocolOptions: LocalProtocolOptions }): Promise { + const localRoot = resolveLocalUrlToPath("local://", options.localProtocolOptions); + try { + const entries = await fs.promises.readdir(localRoot, { withFileTypes: true }); + const plans = await Promise.all( + entries + .filter(entry => entry.isFile() && /plan\.md$/i.test(entry.name)) + .map(async entry => { + const stat = await fs.promises.stat(path.join(localRoot, entry.name)).catch(() => null); + return { url: `local://${entry.name}`, mtime: stat?.mtimeMs ?? 0 }; + }), + ); + return plans.sort((a, b) => b.mtime - a.mtime).map(plan => plan.url); + } catch { + return []; + } +} diff --git a/packages/coding-agent/src/session/acp-permission-gate.ts b/packages/coding-agent/src/session/acp-permission-gate.ts new file mode 100644 index 000000000..fd8fac721 --- /dev/null +++ b/packages/coding-agent/src/session/acp-permission-gate.ts @@ -0,0 +1,170 @@ +import { Patch } from "@oh-my-pi/hashline"; +import { isRecord, stringProperty } from "@oh-my-pi/pi-utils"; +import { expandApplyPatchToEntries } from "../edit"; +import { resolveToCwd } from "../tools/path-utils"; +import type { ClientBridgePermissionOption } from "./client-bridge"; + +/** Tools that require user permission before execution when an ACP client is connected. */ +export const PERMISSION_REQUIRED_TOOLS: Record = { + bash: true, + edit: true, + delete: true, + move: true, +}; + +/** Permission options indexed by their wire identifiers. */ +export const PERMISSION_OPTIONS_BY_ID: Record = { + allow_once: { optionId: "allow_once", name: "Allow once", kind: "allow_once" }, + allow_always: { optionId: "allow_always", name: "Always allow", kind: "allow_always" }, + reject_once: { optionId: "reject_once", name: "Reject", kind: "reject_once" }, + reject_always: { optionId: "reject_always", name: "Always reject", kind: "reject_always" }, +}; + +/** Permission options presented to the client on each gated tool call. */ +export const PERMISSION_OPTIONS: ClientBridgePermissionOption[] = [ + PERMISSION_OPTIONS_BY_ID.allow_once, + PERMISSION_OPTIONS_BY_ID.allow_always, + PERMISSION_OPTIONS_BY_ID.reject_once, + PERMISSION_OPTIONS_BY_ID.reject_always, +]; + +function getEditDestructiveIntent(args: unknown): { kind: "delete" | "move"; paths: string[] } | undefined { + if (!isRecord(args)) return undefined; + + const edits = Array.isArray(args.edits) ? args.edits : undefined; + if (edits) { + const filePath = stringProperty(args, "path"); + if (filePath) { + for (const edit of edits) { + if (!isRecord(edit)) continue; + if (stringProperty(edit, "op") === "delete") return { kind: "delete", paths: [filePath] }; + } + } + for (const edit of edits) { + if (!isRecord(edit)) continue; + const op = stringProperty(edit, "op"); + const rename = stringProperty(edit, "rename"); + if (op !== "create" && rename) return { kind: "move", paths: filePath ? [filePath, rename] : [rename] }; + } + } + + const input = stringProperty(args, "input"); + if (input) { + try { + const patch = Patch.parse(input); + for (const section of patch.sections) { + if (section.fileOp?.kind === "rem") return { kind: "delete", paths: [section.path] }; + if (section.fileOp?.kind === "move") return { kind: "move", paths: [section.path, section.fileOp.dest] }; + } + } catch { + // Not a hashline patch — fall through to apply_patch parsing. + } + try { + const entries = expandApplyPatchToEntries({ input }); + const deleteEntry = entries.find(entry => entry.op === "delete"); + if (deleteEntry) return { kind: "delete", paths: [deleteEntry.path] }; + const moveEntry = entries.find(entry => entry.rename); + if (moveEntry?.rename) return { kind: "move", paths: [moveEntry.path, moveEntry.rename] }; + } catch { + // If the edit input is not an apply-patch envelope, it is not a delete/move operation. + } + } + + return undefined; +} + +/** Describes the permission prompt required for a destructive tool call. */ +export function getPermissionIntent( + toolName: string, + args: unknown, +): { toolName: string; title: string; paths?: string[]; cacheKey: string } | undefined { + const input = isRecord(args) ? args : {}; + if (toolName === "bash") { + const command = stringProperty(input, "command")?.slice(0, 80); + return { toolName, title: command || toolName, cacheKey: toolName }; + } + if (toolName === "delete") { + const filePath = stringProperty(input, "path"); + return { + toolName, + title: filePath ? `Delete ${filePath}` : toolName, + paths: filePath ? [filePath] : undefined, + cacheKey: toolName, + }; + } + if (toolName === "move") { + const from = stringProperty(input, "oldPath") ?? stringProperty(input, "path") ?? stringProperty(input, "from"); + const to = + stringProperty(input, "newPath") ?? stringProperty(input, "to") ?? stringProperty(input, "destination"); + if (from && to) return { toolName, title: `Move ${from} to ${to}`, paths: [from, to], cacheKey: toolName }; + return { + toolName, + title: from ? `Move ${from}` : toolName, + paths: from ? [from] : undefined, + cacheKey: toolName, + }; + } + if (toolName === "edit") { + const intent = getEditDestructiveIntent(args); + if (!intent) return undefined; + if (intent.kind === "delete") { + return { + toolName, + title: `Delete ${intent.paths[0] ?? "edit target"}`, + paths: intent.paths, + cacheKey: "edit:delete", + }; + } + const from = intent.paths[0]; + const to = intent.paths[1]; + return { + toolName, + title: from && to ? `Move ${from} to ${to}` : `Move ${from ?? to ?? "edit target"}`, + paths: intent.paths, + cacheKey: "edit:move", + }; + } + return undefined; +} + +/** Converts tool path arguments into absolute ACP editor locations. */ +export function extractPermissionLocations( + args: unknown, + cwd: string, + explicitPaths?: string[], +): { path: string; line?: number }[] { + if (!isRecord(args)) return []; + const out: { path: string; line?: number }[] = []; + const pushPath = (value: unknown) => { + if (typeof value !== "string" || value.length === 0) return; + // ACP locations carry file paths that the editor host will open or focus; + // they must be absolute or the client cannot resolve them. Resolve raw + // tool args (often cwd-relative) against the session cwd before sending. + let resolved: string; + try { + resolved = resolveToCwd(value, cwd); + } catch { + return; + } + if (out.some(location => location.path === resolved)) return; + out.push({ path: resolved }); + }; + if (explicitPaths) { + for (const filePath of explicitPaths) pushPath(filePath); + return out; + } + pushPath(args.path); + pushPath(args.file); + if (Array.isArray(args.paths)) { + for (const filePath of args.paths) { + if (typeof filePath === "string") pushPath(filePath); + } + } + pushPath(args.oldPath); + pushPath(args.newPath); + pushPath(args.from); + pushPath(args.to); + pushPath(args.source); + pushPath(args.destination); + return out; +} diff --git a/packages/coding-agent/src/session/agent-session-error-log.test.ts b/packages/coding-agent/src/session/agent-session-error-log.test.ts index b68871827..6a5c42716 100644 --- a/packages/coding-agent/src/session/agent-session-error-log.test.ts +++ b/packages/coding-agent/src/session/agent-session-error-log.test.ts @@ -7,7 +7,7 @@ import { afterAll, afterEach, describe, expect, it } from "bun:test"; import type { AssistantMessage } from "@oh-my-pi/pi-ai"; import { logger } from "@oh-my-pi/pi-utils"; -import { logProviderTurnError } from "./agent-session"; +import { logProviderTurnError } from "./messages"; function makeMessage(overrides: Partial): AssistantMessage { return { diff --git a/packages/coding-agent/src/session/agent-session-events.ts b/packages/coding-agent/src/session/agent-session-events.ts new file mode 100644 index 000000000..72c0ce0bb --- /dev/null +++ b/packages/coding-agent/src/session/agent-session-events.ts @@ -0,0 +1,66 @@ +import type { AgentEvent, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import type { CompactionResult } from "@oh-my-pi/pi-agent-core/compaction"; +import type { Effort } from "@oh-my-pi/pi-ai"; +import type { Rule } from "../capability/rule"; +import type { RecoveredRetryError } from "../extensibility/shared-events"; +import type { Goal, GoalModeState } from "../goals/state"; +import type { ConfiguredThinkingLevel } from "../thinking"; +import type { TodoItem } from "../tools/todo"; +import type { CustomMessage } from "./messages"; + +/** Session-specific events that extend the core AgentEvent. */ +export type AgentSessionEvent = + | Exclude + | (Extract & { + /** False when an async delivery will resume the session before its true final settle. */ + isTerminal?: boolean; + }) + | { + type: "auto_compaction_start"; + reason: "threshold" | "overflow" | "idle" | "incomplete"; + action: "context-full" | "handoff" | "shake" | "snapcompact"; + } + | { + type: "auto_compaction_end"; + action: "context-full" | "handoff" | "shake" | "snapcompact"; + result: CompactionResult | undefined; + aborted: boolean; + willRetry: boolean; + errorMessage?: string; + /** True when compaction was skipped for a benign reason. */ + skipped?: boolean; + } + | { + type: "auto_retry_start"; + attempt: number; + maxAttempts: number; + delayMs: number; + errorMessage: string; + errorId?: number; + } + | { + type: "auto_retry_end"; + success: boolean; + attempt: number; + finalError?: string; + recoveredErrors?: RecoveredRetryError[]; + } + | { type: "retry_fallback_applied"; from: string; to: string; role: string } + | { type: "retry_fallback_succeeded"; model: string; role: string } + | { type: "ttsr_triggered"; rules: Rule[] } + | { type: "todo_reminder"; todos: TodoItem[]; attempt: number; maxAttempts: number } + | { type: "todo_auto_clear" } + | { type: "irc_message"; message: CustomMessage } + | { type: "notice"; level: "info" | "warning" | "error"; message: string; source?: string } + | { + type: "thinking_level_changed"; + thinkingLevel: ThinkingLevel | undefined; + /** The user-configured selector when it differs from the effective level. */ + configured?: ConfiguredThinkingLevel; + /** The level `auto` resolved to this turn, once classified. */ + resolved?: Effort; + } + | { type: "goal_updated"; goal: Goal | null; state?: GoalModeState }; + +/** Listener function for agent session events. */ +export type AgentSessionEventListener = (event: AgentSessionEvent) => void; diff --git a/packages/coding-agent/src/session/agent-session-types.ts b/packages/coding-agent/src/session/agent-session-types.ts new file mode 100644 index 000000000..ad31d7127 --- /dev/null +++ b/packages/coding-agent/src/session/agent-session-types.ts @@ -0,0 +1,325 @@ +import type { Agent, AgentMessage, AgentTool, StreamFn, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import type { + Context, + ImageContent, + Message, + MessageAttribution, + Model, + ServiceTierByFamily, + SimpleStreamOptions, + ToolChoice, +} from "@oh-my-pi/pi-ai"; +import type { postmortem } from "@oh-my-pi/pi-utils"; +import type { AdvisorConfig } from "../advisor"; +import type { AsyncJob, AsyncJobDeliveryState, AsyncJobManager } from "../async"; +import type { ModelRegistry } from "../config/model-registry"; +import type { PromptTemplate } from "../config/prompt-templates"; +import type { Settings, SkillsSettings } from "../config/settings"; +import type { RawSseDebugBuffer } from "../debug/raw-sse-buffer"; +import type { TtsrManager } from "../export/ttsr"; +import type { LoadedCustomCommand } from "../extensibility/custom-commands"; +import type { ExtensionRunner } from "../extensibility/extensions"; +import type { ContextUsage } from "../extensibility/extensions/types"; +import type { Skill, SkillWarning } from "../extensibility/skills"; +import type { FileSlashCommand } from "../extensibility/slash-commands"; +import type { SecretObfuscator } from "../secrets/obfuscator"; +import type { ConfiguredThinkingLevel } from "../thinking"; +import type { XdevRegistry } from "../tools/xdev"; +import type { SessionManager } from "./session-manager"; + +/** Maximum time the interactive shutdown path waits for Mnemopi consolidation. */ +export const SHUTDOWN_CONSOLIDATE_BUDGET_MS = 1_500; + +/** Options controlling session disposal. */ +export interface AgentSessionDisposeOptions { + mnemopiConsolidateTimeoutMs?: number; + /** + * Postmortem reason that triggered this dispose (signal/fatal teardown + * paths). When set, the persisted `session_exit` diagnostic records it + * instead of the generic `"dispose"` used for normal programmatic disposal + * (`/quit`, test teardown, subagent completion). + */ + reason?: postmortem.Reason; +} + +/** Listener notified when command metadata changes. */ +export type CommandMetadataChangedListener = () => void | Promise; +/** Public summary of an asynchronous job. */ +export type AsyncJobSnapshotItem = Pick; + +/** Snapshot of running, recent, and pending-delivery asynchronous jobs. */ +export interface AsyncJobSnapshot { + running: AsyncJobSnapshotItem[]; + recent: AsyncJobSnapshotItem[]; + delivery: AsyncJobDeliveryState; +} + +export type { ShakeMode, ShakeResult } from "./shake-types"; + +/** + * Prewalk switches an active session one-way from its starting model to a + * fast/cheap target after implementation begins. + */ +export interface Prewalk { + target: Model; + thinkingLevel?: ConfiguredThinkingLevel; +} + +/** + * PlanYolo starts in read-only plan mode, auto-approves the proposal, then + * switches to a target model for implementation. + */ +export interface PlanYolo { + target: Model; + thinkingLevel?: ConfiguredThinkingLevel; +} + +/** Identifies a retry fallback chain already entered during startup model resolution. */ +export interface InitialRetryFallbackState { + /** Role whose configured primary was unavailable. */ + role: string; + /** Configured primary selector retained for restoration when it becomes available. */ + originalSelector: string; + /** Thinking selector configured for the unavailable primary. */ + originalThinkingLevel: ConfiguredThinkingLevel | undefined; +} + +/** Dependencies and initial state used to construct an AgentSession. */ +export interface AgentSessionConfig { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + /** Whether the caller explicitly requested yolo/auto-approve behavior for this session. */ + autoApprove?: boolean; + /** Models to cycle through with Ctrl+P (from --models flag). */ + scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; + /** Initial session thinking selector. */ + thinkingLevel?: ConfiguredThinkingLevel; + /** Retry chain ownership when startup selected one of its fallback entries. */ + initialRetryFallback?: InitialRetryFallbackState; + /** Prewalk from the starting model to a fast/cheap target after implementation begins. */ + prewalk?: Prewalk; + /** Force read-only plan mode at start, auto-approve, then switch to the target. */ + planYolo?: PlanYolo; + /** Initial per-family service tiers for the live session. */ + serviceTierByFamily?: ServiceTierByFamily; + /** Prompt templates for expansion. */ + promptTemplates?: PromptTemplate[]; + /** File-based slash commands for expansion. */ + slashCommands?: FileSlashCommand[]; + /** Extension runner created with wrapped tools. */ + extensionRunner?: ExtensionRunner; + /** Loaded skills already discovered by the SDK. */ + skills?: Skill[]; + /** Skill loading warnings already captured by the SDK. */ + skillWarnings?: SkillWarning[]; + /** Whether runtime reloads may rediscover disk-backed skills. */ + skillsReloadable?: boolean; + /** Custom TypeScript slash commands. */ + customCommands?: LoadedCustomCommand[]; + skillsSettings?: SkillsSettings; + /** Agent directory used when changing memory backends in a live session. */ + memoryAgentDir?: string; + /** Recursion depth used to suppress live backend replacement in subagents. */ + memoryTaskDepth?: number; + /** Creates built-in memory tools for the current backend. */ + createMemoryTools?: () => Promise; + /** Model registry for API key resolution and model discovery. */ + modelRegistry: ModelRegistry; + /** Tool registry for LSP and settings. */ + toolRegistry?: Map; + /** Creates tools registered only while vibe mode is active. */ + createVibeTools?: () => AgentTool[]; + /** Names whose current registry entry is the built-in implementation. */ + builtInToolNames?: Iterable; + /** Updates tool-session predicates from the live active tool set. */ + setActiveToolNames?: (names: Iterable) => void; + /** Registers the write transport when runtime xdev mounts first need it. */ + ensureWriteRegistered?: () => Promise; + /** Current session pre-LLM message transform pipeline. */ + transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise; + /** Provider request transform applied after message conversion. */ + transformProviderContext?: (context: Context, model: Model) => Context | Promise; + /** Stream wrapper for side-channel requests. */ + sideStreamFn?: StreamFn; + /** Stream wrapper for advisor requests. */ + advisorStreamFn?: StreamFn; + /** Prefer websocket transport for OpenAI Codex requests when supported. */ + preferWebsockets?: boolean; + /** Provider payload hook used by the active session request path. */ + onPayload?: SimpleStreamOptions["onPayload"]; + /** Provider response hook used by the active session request path. */ + onResponse?: SimpleStreamOptions["onResponse"]; + /** Raw SSE hook used by the active session request path. */ + onSseEvent?: SimpleStreamOptions["onSseEvent"]; + /** Per-session raw SSE diagnostic buffer. */ + rawSseDebugBuffer?: RawSseDebugBuffer; + /** Current session message-to-LLM conversion pipeline. */ + convertToLlm?: (messages: AgentMessage[]) => Message[] | Promise; + /** System prompt builder that can consider tool availability. */ + rebuildSystemPrompt?: (toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>; + /** Local calendar date provider used by prompt-cache invalidation. */ + getLocalCalendarDate?: () => string; + /** Tools mounted under `xd://`, for `/tools` display. */ + getXdevToolEntries?: () => Array<{ name: string; summary: string }>; + /** Session-owned `xd://` registry. */ + xdevRegistry?: XdevRegistry; + /** Discoverable tools mounted under `xd://` in the initial enabled set. */ + initialMountedXdevToolNames?: string[]; + /** Names pinned top-level during runtime repartitioning. */ + presentationPinnedToolNames?: ReadonlySet; + /** Accessor for live MCP server instructions. */ + getMcpServerInstructions?: () => Map | undefined; + /** Time-traveling stream-rule manager. */ + ttsrManager?: TtsrManager; + /** Secret obfuscator for provider and edit content. */ + obfuscator?: SecretObfuscator; + /** Inherited eval executor session id from a parent agent. */ + parentEvalSessionId?: string; + /** Logical owner for retained eval kernels created by this session. */ + evalKernelOwnerId?: string; + /** Async job manager owned and disposed by this session. */ + ownedAsyncJobManager?: AsyncJobManager; + /** Async job manager visible to this session. */ + asyncJobManager?: AsyncJobManager; + /** Registry identity used for IRC routing. */ + agentId?: string; + /** Whether this is a top-level or subagent session. */ + agentKind?: "main" | "sub"; + /** Provider-facing session ID override. */ + providerSessionId?: string; + /** Whether the provider prompt-cache key was explicit or fork-inherited. */ + providerPromptCacheKeySource?: "explicit" | "fork"; + /** Full advisor toolset built against an advisor-scoped tool session. */ + advisorTools?: AgentTool[]; + /** Preloaded watchdog prompt content for the advisor. */ + advisorWatchdogPrompt?: string; + /** Shared advisor instructions loaded from WATCHDOG.yml. */ + advisorSharedInstructions?: string; + /** Project context rendered for advisor sessions. */ + advisorContextPrompt?: string; + /** Advisors discovered from WATCHDOG.yml. */ + advisorConfigs?: AdvisorConfig[]; + /** Strip tool descriptions from provider-bound side-request tool specs. */ + pruneToolDescriptions?: boolean; + /** Disconnect the MCP manager owned by this session during disposal. */ + disconnectOwnedMcpManager?: () => Promise; + /** System prompt used by automatic session-title generation. */ + titleSystemPrompt?: string; +} + +/** Options for AgentSession.prompt(). */ +export interface PromptOptions { + /** Whether to expand file-based prompt templates (default: true). */ + expandPromptTemplates?: boolean; + /** Image attachments. */ + images?: ImageContent[]; + /** Queue behavior while streaming. */ + streamingBehavior?: "steer" | "followUp"; + /** Optional tool choice override for the next LLM call. */ + toolChoice?: ToolChoice; + /** Send as a developer/system message instead of user. */ + synthetic?: boolean; + /** Whether this prompt is a deliberate user action. */ + userInitiated?: boolean; + /** Explicit billing/initiator attribution. */ + attribution?: MessageAttribution; + /** Skip pre-send compaction checks for this prompt. */ + skipCompactionCheck?: boolean; +} + +/** Options for AgentSession.followUp(). */ +export interface FollowUpOptions { + /** Enqueue as a hidden developer message instead of a user follow-up. */ + synthetic?: boolean; + /** Whether to expand file-based prompt templates (default: true). */ + expandPromptTemplates?: boolean; + /** Explicit billing/initiator attribution. */ + attribution?: MessageAttribution; +} + +/** Result from a handoff operation. */ +export interface HandoffResult { + document: string; + savedPath?: string; +} + +/** Options controlling handoff generation. */ +export interface SessionHandoffOptions { + autoTriggered?: boolean; + signal?: AbortSignal; + onSwitchCancelled?: () => void; +} + +/** Result from cycleModel(). */ +export interface ModelCycleResult { + model: Model; + thinkingLevel: ThinkingLevel | undefined; + /** Whether cycling through scoped models or all available models. */ + isScoped: boolean; +} + +/** Result from cycleRoleModels(). */ +export interface RoleModelCycleResult { + model: Model; + thinkingLevel: ThinkingLevel | undefined; + role: string; +} + +/** A configured role resolved to a concrete model. */ +export interface ResolvedRoleModel { + role: string; + model: Model; + thinkingLevel?: ConfiguredThinkingLevel; + explicitThinkingLevel: boolean; +} + +/** Resolvable role models and the currently active index. */ +export interface RoleModelCycle { + models: ResolvedRoleModel[]; + currentIndex: number; +} + +/** Token breakdown for the current provider context. */ +export interface ContextUsageBreakdown { + contextWindow: number; + anchored: boolean; + usedTokens: number; + systemPromptTokens: number; + systemToolsTokens: number; + systemContextTokens: number; + skillsTokens: number; + messagesTokens: number; +} + +/** Session statistics for the `/session` command. */ +export interface SessionStats { + sessionFile: string | undefined; + sessionId: string; + userMessages: number; + assistantMessages: number; + toolCalls: number; + toolResults: number; + totalMessages: number; + tokens: { + input: number; + output: number; + reasoning: number; + cacheRead: number; + cacheWrite: number; + total: number; + }; + premiumRequests: number; + cost: number; + contextUsage?: ContextUsage; +} + +/** IDs for a newly created session and the session it replaced. */ +export interface FreshSessionResult { + previousSessionId: string; + sessionId: string; + closedProviderSessions: number; +} + +/** Queued user content restored to the editor. */ +export type RestoredQueuedMessage = { text: string; images?: ImageContent[] }; diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 1ec5a9fad..1e7e1e49b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -20,11 +20,10 @@ import { scheduler } from "node:timers/promises"; import { isPromise } from "node:util/types"; import type { InMemorySnapshotStore } from "@oh-my-pi/hashline"; -import { Patch } from "@oh-my-pi/hashline"; import { type AfterToolCallContext, type AfterToolCallResult, - Agent, + type Agent, AgentBusyError, type AgentEvent, type AgentMessage, @@ -36,67 +35,27 @@ import { type AgentTurnEndContext, AppendOnlyContextManager, type AsideMessage, - type CompactionSummaryMessage, - countTokens, - createToolScopedAbortReason, - isSyntheticToolResultMessage, resolveTelemetry, type StreamFn, TERMINAL_TOOL_RESULT_ABORT_REASON, - ThinkingLevel, + type ThinkingLevel, type ToolChoiceDirective, } from "@oh-my-pi/pi-agent-core"; import { - AGGRESSIVE_SHAKE_CONFIG, - AUTO_HANDOFF_THRESHOLD_FOCUS, - applyShakeRegions, - CompactionCancelledError, type CompactionPreparation, type CompactionResult, - type CompactionSettings, - calculateContextTokens, calculatePromptTokens, collectEntriesForBranchSummary, - collectShakeRegions, - compact, - compactionContextTokens, - createCompactionSummaryMessage, - DEFAULT_SHAKE_CONFIG, - effectiveReserveTokens, estimateTokens, generateBranchSummary, - generateHandoffFromContext, - invalidateMessageCache, - prepareCompaction, - renderHandoffPrompt, - resolveBudgetReserveTokens, - resolveThresholdTokens, - type SessionMessageEntry, type ShakeConfig, - type ShakeRegion, - type SummaryOptions, - shouldCompact, - shouldUseOpenAiRemoteCompaction, } from "@oh-my-pi/pi-agent-core/compaction"; -import { - DEFAULT_PRUNE_CONFIG, - pruneSupersededToolResults, - pruneToolOutputs, - readToolSupersedeKey, -} from "@oh-my-pi/pi-agent-core/compaction/pruning"; -import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection"; import type { AssistantMessage, - AssistantMessageEvent, - AssistantRetryRecovery, - AssistantRetryRecoveryKind, CodexCompactionContext, - Context, ImageContent, Message, - MessageAttribution, Model, - ProviderResponseMetadata, ProviderSessionState, ResetCreditAccountStatus, ResetCreditRedeemOutcome, @@ -109,121 +68,46 @@ import type { ToolCall, ToolChoice, ToolResultMessage, - Usage, UsageReport, } from "@oh-my-pi/pi-ai"; -import { - calculateRateLimitBackoffMs, - clearAnthropicFastModeFallback, - deriveClaudeDeviceId, - Effort, - isUsageLimitOutcome, - parseRateLimitReason, - realizesPriorityServiceTier, - resolveModelServiceTier, - serviceTierFamily, - streamSimple, -} from "@oh-my-pi/pi-ai"; +import { deriveClaudeDeviceId, type Effort, streamSimple } from "@oh-my-pi/pi-ai"; import * as AIError from "@oh-my-pi/pi-ai/error"; import { resetOpenAICodexHistoryAfterCompaction } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; -import { kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; import { toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema"; -import { GeminiHeaderRunDetector, isGeminiThinkingModel } from "@oh-my-pi/pi-ai/utils/thinking-loop"; -import { type RepeatedToolCallDetection, ToolCallLoopGuard } from "@oh-my-pi/pi-ai/utils/tool-call-loop-guard"; -import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; -import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; -import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives"; import { escapeXmlText, - extractHttpStatusFromError, - extractRetryHint, formatDuration, getAgentDbPath, getInstallId, isBunTestRuntime, isEnoent, isInteractiveHost, + isRecord, logger, postmortem, prompt, - relativePathWithinRoot, Snowflake, + stringProperty, withTimeout, } from "@oh-my-pi/pi-utils"; -import * as snapcompact from "@oh-my-pi/snapcompact"; -import { - ADVISOR_DEFAULT_TOOL_NAMES, - AdviseTool, - type AdvisorAgent, - type AdvisorConfig, - AdvisorEmissionGuard, - type AdvisorMessageDetails, - type AdvisorNote, - AdvisorOutputQuarantinedError, - AdvisorRuntime, - type AdvisorRuntimeStatus, - type AdvisorSeverity, - AdvisorTranscriptRecorder, - advisorTranscriptFilename, - annotateForStaleness, - buildAdvisorQuarantineSourceText, - formatAdvisorBatchContent, - getOrCreateAdvisorProviderSessionId, - isAdvisorInterruptImmuneTurnActive, - isInterruptingSeverity, - quarantineAdvisorUnsafeOutput, - resolveAdvisorDeliveryChannel, - slugifyAdvisorName, -} from "../advisor"; -import { type AsyncJob, type AsyncJobDeliveryState, AsyncJobManager } from "../async"; -import { classifyDifficulty } from "../auto-thinking/classifier"; -import { reset as resetCapabilities } from "../capability"; -import type { Rule } from "../capability/rule"; +import type { AdvisorConfig, AdvisorRuntimeStatus } from "../advisor"; +import { type AsyncJob, AsyncJobManager } from "../async"; import { shouldEnableAppendOnlyContext } from "../config/append-only-context-mode"; import type { ModelRegistry } from "../config/model-registry"; -import { - extractExplicitThinkingSelector, - filterAvailableModelsByEnabledPatterns, - formatModelSelectorValue, - formatModelString, - formatModelStringWithRouting, - getModelMatchPreferences, - parseModelString, - type ResolvedModelRoleValue, - resolveAdvisorRoleSelection, - resolveModelOverride, - resolveModelRoleValue, -} from "../config/model-resolver"; -import { getKnownRoleIds, MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles"; +import type { ResolvedModelRoleValue } from "../config/model-resolver"; import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates"; -import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; +import { buildServiceTierByFamily } from "../config/service-tier"; import type { Settings, SkillsSettings } from "../config/settings"; -import { - getDefault, - onAppendOnlyModeChanged, - onModelRolesChanged, - validateProviderMaxInFlightRequests, -} from "../config/settings"; -import { CursorExecHandlers } from "../cursor"; +import { onAppendOnlyModeChanged, onModelRolesChanged } from "../config/settings"; import { RawSseDebugBuffer } from "../debug/raw-sse-buffer"; -import { expandApplyPatchToEntries, normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit"; import { getFileSnapshotStore } from "../edit/file-snapshot-store"; -import { disposeJuliaKernelSessionsByOwner } from "../eval/jl/executor"; -import { namespaceSessionId as namespacePythonSessionId } from "../eval/py"; -import { - disposeKernelSessionsByOwner, - executePython as executePythonCommand, - type PythonResult, -} from "../eval/py/executor"; -import { disposeRubyKernelSessionsByOwner } from "../eval/rb/executor"; -import { defaultEvalSessionId } from "../eval/session-id"; -import { type BashResult, executeBash as executeBashCommand } from "../exec/bash-executor"; -import type { TtsrManager, TtsrMatchContext } from "../export/ttsr"; +import type { PythonResult } from "../eval/py/executor"; +import type { BashResult } from "../exec/bash-executor"; +import type { TtsrManager } from "../export/ttsr"; import type { LoadedCustomCommand } from "../extensibility/custom-commands"; -import type { CustomTool, CustomToolContext } from "../extensibility/custom-tools/types"; -import { CustomToolAdapter } from "../extensibility/custom-tools/wrapper"; +import type { CustomTool } from "../extensibility/custom-tools/types"; import type { ExtensionCommandContext, ExtensionRunner, @@ -232,7 +116,6 @@ import type { MessageStartEvent, MessageUpdateEvent, SessionBeforeBranchResult, - SessionBeforeCompactResult, SessionBeforeSwitchResult, SessionBeforeTreeResult, SessionStopEventResult, @@ -247,95 +130,56 @@ import { emitSessionShutdownEvent } from "../extensibility/extensions"; import { ManagedTimers } from "../extensibility/extensions/managed-timers"; import { createExtensionModelQuery } from "../extensibility/extensions/model-api"; import type { CompactOptions, ContextUsage } from "../extensibility/extensions/types"; -import { ExtensionToolWrapper } from "../extensibility/extensions/wrapper"; import type { HookCommandContext } from "../extensibility/hooks/types"; -import type { RecoveredRetryError } from "../extensibility/shared-events"; -import { loadSkills, type Skill, type SkillWarning, setActiveSkills } from "../extensibility/skills"; +import type { Skill, SkillWarning } from "../extensibility/skills"; import { expandSlashCommand, type FileSlashCommand } from "../extensibility/slash-commands"; import { GoalRuntime } from "../goals/runtime"; -import type { Goal, GoalModeState } from "../goals/state"; +import type { GoalModeState } from "../goals/state"; import type { HindsightSessionState } from "../hindsight/state"; import { type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls"; -import { IrcBus, type IrcMessage } from "../irc/bus"; -import { resolveMemoryBackend } from "../memory-backend/resolve"; -import { MEMORY_BACKEND_TOOL_NAMES } from "../memory-backend/tool-names"; +import type { IrcMessage } from "../irc/bus"; import { shutdownMnemopiEmbedClient } from "../mnemopi/embed-client"; import { getMnemopiSessionState, type MnemopiSessionState, setMnemopiSessionState } from "../mnemopi/state"; import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate"; import { theme } from "../modes/theme/theme"; import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; -import { - computeNonMessageBreakdown, - computeNonMessageTokens, - estimateToolSchemaTokens, -} from "../modes/utils/context-usage"; +import { computeNonMessageTokens } from "../modes/utils/context-usage"; import { containsWorkflow, renderWorkflowNotice } from "../modes/workflow"; import { type PlanApprovalDetails, resolveApprovedPlan } from "../plan-mode/approved-plan"; -import { createPlanReadMatcher } from "../plan-mode/plan-protection"; +import { listPlanFiles, readPlanFile } from "../plan-mode/plan-files"; import type { PlanModeState } from "../plan-mode/state"; -import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" }; import goalModeContextPrompt from "../prompts/goals/goal-mode-context.md" with { type: "text" }; import goalTodoContextPrompt from "../prompts/goals/goal-todo-context.md" with { type: "text" }; -import parentIrcSteerTemplate from "../prompts/steering/parent-irc.md" with { type: "text" }; import autoContinuePrompt from "../prompts/system/auto-continue.md" with { type: "text" }; -import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" }; -import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" }; -import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" }; -import geminiToolReminderTemplate from "../prompts/system/gemini-tool-call-reminder.md" with { type: "text" }; import interruptedThinkingTemplate from "../prompts/system/interrupted-thinking.md" with { type: "text" }; -import ircAutoReplyTemplate from "../prompts/system/irc-autoreply.md" with { type: "text" }; -import ircIncomingTemplate from "../prompts/system/irc-incoming.md" with { type: "text" }; -import midRunTodoNudgePrompt from "../prompts/system/mid-run-todo-nudge.md" with { type: "text" }; import planModeActivePrompt from "../prompts/system/plan-mode-active.md" with { type: "text" }; import planModeReferencePrompt from "../prompts/system/plan-mode-reference.md" with { type: "text" }; import planModeToolDecisionReminderPrompt from "../prompts/system/plan-mode-tool-decision-reminder.md" with { type: "text", }; -import planYoloHandoffPrompt from "../prompts/system/plan-yolo-handoff.md" with { type: "text" }; -import prewalkChecklistPrompt from "../prompts/system/prewalk-checklist.md" with { type: "text" }; -import prewalkContinuePrompt from "../prompts/system/prewalk-continue.md" with { type: "text" }; -import prewalkPlanPrompt from "../prompts/system/prewalk-plan.md" with { type: "text" }; import rewindReportTemplate from "../prompts/system/rewind-report.md" with { type: "text" }; import sideChannelNoToolsReminder from "../prompts/system/side-channel-no-tools.md" with { type: "text" }; -import thinkingLoopRedirectTemplate from "../prompts/system/thinking-loop-redirect.md" with { type: "text" }; -import toolCallLoopRedirectTemplate from "../prompts/system/tool-call-loop-redirect.md" with { type: "text" }; -import ttsrInterruptTemplate from "../prompts/system/ttsr-interrupt.md" with { type: "text" }; -import ttsrToolReminderTemplate from "../prompts/system/ttsr-tool-reminder.md" with { type: "text" }; -import unexpectedStopRetryTemplate from "../prompts/system/unexpected-stop-retry.md" with { type: "text" }; import vibeModeActivePrompt from "../prompts/system/vibe-mode-active.md" with { type: "text" }; -import xdevMountNoticePrompt from "../prompts/system/xdev-mount-notice.md" with { type: "text" }; -import { AgentRegistry } from "../registry/agent-registry"; import { deobfuscateAssistantContent, deobfuscateSessionContext, deobfuscateToolArguments, - obfuscateMessages, obfuscateProviderContext, type SecretObfuscator, - stripPendingSecretPlaceholderSuffix, } from "../secrets/obfuscator"; -import { usesCodexTaskPrompt } from "../task/prompt-policy"; import { AUTO_THINKING, type ConfiguredThinkingLevel, - clampAutoThinkingEffort, - concreteThinkingLevel, parseConfiguredThinkingLevel, - resolveProvisionalAutoLevel, - resolveThinkingLevelForModel, shouldDisableReasoning, toReasoningEffort, } from "../thinking"; -import { formatTitleConversationContext, type TitleConversationTurn } from "../tiny/message-preproc"; import { shutdownTinyTitleClient } from "../tiny/title-client"; import { type AskToolDetails, type AskToolInput, recoverAskQuestions } from "../tools/ask"; -import { assertEditableFile } from "../tools/auto-generated-guard"; import { releaseTabsForOwner } from "../tools/browser/tab-supervisor"; -import { isMCPToolName, normalizeToolNames } from "../tools/builtin-names"; import type { CheckpointState, CompletedRewindState } from "../tools/checkpoint"; -import { outputMeta, wrapToolWithMetaNotice } from "../tools/output-meta"; -import { isInternalUrlPath, normalizeLocalScheme, resolveToCwd } from "../tools/path-utils"; +import { normalizeLocalScheme, resolveToCwd } from "../tools/path-utils"; import { buildResolveReminderMessage, isPreviewResolutionToolCall, @@ -344,20 +188,36 @@ import { PROPOSE_DEVICE_NAME, writeDeviceDispatch, } from "../tools/resolve"; -import { getLatestTodoPhasesFromEntries, type TodoItem, type TodoPhase } from "../tools/todo"; -import { ToolAbortError, ToolError } from "../tools/tool-errors"; -import { clampTimeout } from "../tools/tool-timeouts"; -import { isMountableUnderXdev, type XdevRegistry } from "../tools/xdev"; +import type { TodoPhase } from "../tools/todo"; +import { ToolError } from "../tools/tool-errors"; import { parseCommandArgs } from "../utils/command-args"; -import { type EditMode, resolveEditMode } from "../utils/edit-mode"; +import type { EditMode } from "../utils/edit-mode"; import { resolveFileDisplayMode } from "../utils/file-display-mode"; import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions"; import { normalizeModelContextImages } from "../utils/image-loading"; -import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback"; -import { formatLocalCalendarDate } from "../utils/local-date"; import { generateSessionTitle } from "../utils/title-generator"; import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice"; import type { VibeModeState } from "../vibe/state"; +import type { AgentSessionEvent, AgentSessionEventListener } from "./agent-session-events"; +import type { + AgentSessionConfig, + AgentSessionDisposeOptions, + AsyncJobSnapshot, + CommandMetadataChangedListener, + ContextUsageBreakdown, + FollowUpOptions, + FreshSessionResult, + HandoffResult, + ModelCycleResult, + Prewalk, + PromptOptions, + ResolvedRoleModel, + RestoredQueuedMessage, + RoleModelCycle, + RoleModelCycleResult, + SessionHandoffOptions, + SessionStats, +} from "./agent-session-types"; import { ASYNC_INLINE_RESULT_MAX_CHARS, ASYNC_PREVIEW_MAX_CHARS, @@ -366,7 +226,14 @@ import { buildAsyncResultBatchMessage, } from "./async-job-delivery"; import type { AuthStorage } from "./auth-storage"; -import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge"; +import { BashRunner, type BashRunnerHost } from "./bash-runner"; +import { + checkpointStartedAtFromEntry, + completedRewindFromEntry, + isSuccessfulCheckpointEntry, + semanticToolResult, +} from "./checkpoint-entries"; +import type { ClientBridge } from "./client-bridge"; import { type CodexAutoRedeemRedeemDecision, defaultCodexAutoRedeemCoordinator, @@ -374,7 +241,7 @@ import { shouldEvaluateCodexAutoRedeem, shouldPromptCodexAutoRedeem, } from "./codex-auto-reset"; -import { findCompactMode } from "./compact-modes"; +import { EvalRunner, type EvalRunnerHost } from "./eval-runner"; import { collectPendingToolCalls, createInterruptedTurnAbortMessage, @@ -384,931 +251,75 @@ import { TOOL_EXECUTION_START_CUSTOM_TYPE, type ToolExecutionStartData, } from "./exit-diagnostics"; +import { IrcBridge, type IrcBridgeHost } from "./irc-bridge"; import { type BashExecutionMessage, + buildReplanTitleContext, type CustomMessage, type CustomMessagePayload, convertToLlm, + dedupeEphemeralReply, demoteInterruptedThinking, + didSessionMessagesChange, type FileMentionMessage, type HookMessage, INTERRUPTED_THINKING_MESSAGE_TYPE, type InterruptedThinkingDetails, isEmptyErrorTurn, isUserInterruptAbort, + logProviderTurnError, normalizeCustomMessagePayload, type PythonExecutionMessage, - readQueueChipText, SILENT_ABORT_MARKER, SKILL_PROMPT_MESSAGE_TYPE, - stripImagesFromMessage, + sanitizeAssistantForReparentedHistory, USER_INTERRUPT_LABEL, } from "./messages"; +import { ModelControls, type ModelControlsHost } from "./model-controls"; +import { PrewalkCoordinator, type PrewalkCoordinatorHost } from "./prewalk"; +import { + isAdvisorCard, + isDisplayableQueuedMessage, + isHiddenUserCompanion, + isUserQueuedMessage, + queueChipText, + toRestoredQueuedMessage, +} from "./queued-messages"; +import { type AdvisorStats, SessionAdvisors, type SessionAdvisorsHost } from "./session-advisors"; import type { BuildSessionContextOptions, SessionContext } from "./session-context"; -import { getLatestCompactionEntry, getRestorableSessionModels } from "./session-context"; +import { getRestorableSessionModels } from "./session-context"; import { formatSessionDumpText } from "./session-dump-format"; -import type { BranchSummaryEntry, CompactionEntry, NewSessionOptions, SessionEntry } from "./session-entries"; -import { EPHEMERAL_MODEL_CHANGE_ROLE } from "./session-entries"; -import { formatSessionHistoryMarkdown } from "./session-history-format"; +import type { BranchSummaryEntry, NewSessionOptions } from "./session-entries"; +import { SessionHandoff, type SessionHandoffHost } from "./session-handoff"; +import { + COMPACTION_CHECK_NONE, + createCodexCompactionContext as createMaintenanceCodexCompactionContext, + SessionMaintenance, + type SessionMaintenanceHost, +} from "./session-maintenance"; import { cleanupEmptyMoveSession, type SessionManager } from "./session-manager"; +import { SessionMemory, type SessionMemoryHost } from "./session-memory"; +import { SessionProviderBoundary, type SessionProviderBoundaryHost } from "./session-provider-boundary"; +import { SessionStatsTracker, type SessionStatsTrackerHost } from "./session-stats"; +import { SessionTools, type SessionToolsHost } from "./session-tools"; import type { ShakeMode, ShakeResult } from "./shake-types"; import { ToolChoiceQueue } from "./tool-choice-queue"; import { planTurnPersistence, sameMessageContent, sessionMessagePersistenceKey } from "./turn-persistence"; -import { classifyUnexpectedStop, isUnexpectedStopCandidate } from "./unexpected-stop-classifier"; +import { TurnRecovery, type TurnRecoveryHost } from "./turn-recovery"; import { YieldQueue } from "./yield-queue"; +export * from "./agent-session-events"; +export * from "./agent-session-types"; +export * from "./session-advisors"; + const SESSION_STOP_CONTINUATION_CAP = 8; + +import { LoopGuards, type StreamGuardsHost, StreamingEditGuard } from "./stream-guards"; +import { TodoTracker, type TodoTrackerHost } from "./todo-tracker"; +import { TtsrCoordinator, type TtsrCoordinatorHost } from "./ttsr-coordinator"; + const PLAN_MODE_REMINDER_MAX = 3; -type BashAppendDestination = - | { kind: "current"; manager: SessionManager } - | { kind: "detached"; manager: SessionManager } - | { kind: "branch"; manager: SessionManager; parentId: string | null }; - -interface BashSessionTarget { - sessionId: string; - refs: number; - destination?: BashAppendDestination; - pending?: Promise; -} - -interface PendingBashMessage { - target: BashSessionTarget; - message: BashExecutionMessage; -} - -interface BashSessionTransition { - oldTarget: BashSessionTarget; - newTarget: BashSessionTarget; - oldSessionId: string; - oldSessionFile: string | undefined; - oldLeafId: string | null; - detachedManager: SessionManager | undefined; - resolveOld: ((destination: BashAppendDestination) => void) | undefined; - resolveNew: (destination: BashAppendDestination) => void; -} - -/** - * Mutating tool results (`bash`/`eval`/`edit`/`write`/`ast_edit`) without the - * agent touching the `todo` tool that trip the mid-run reconciliation nudge. - * Read-only exploration (grep/read/glob/lsp) never ticks this: an agent - * researching for a long stretch has nothing to flip. Picked so a normal - * fix-verify loop (~3-6 mutations) never sees the nudge, but a sustained run - * of landed work without flipping any todos does. Without this nudge, long - * runs drive the live todo HUD to `0/N` until the final stop, then batch-flip - * to `N/N` (issue #3651). - */ -const MID_RUN_TODO_NUDGE_MUTATION_THRESHOLD = 12; -/** Mid-run nudges per prompt cycle. Deliberately tighter than - * `todo.remindersMax` (the stop-time budget): this is a gentle hidden hint, - * not an escalation ladder. */ -const MID_RUN_TODO_NUDGE_MAX_PER_CYCLE = 2; -/** Tool results that count as landed work for the mid-run todo nudge. */ -const MID_RUN_TODO_NUDGE_MUTATING_TOOLS: Record = { - bash: true, - eval: true, - edit: true, - write: true, - ast_edit: true, -}; -const MARKDOWN_PROMPT_PREFIX_RE = /^(?:>\s*)?(?:(?:[-*+]|\d+[.)])\s+)*/; -const PROMPT_LABEL_RE = /^(?:q(?:uestion)?|ask)\s*\d*\s*[:.)-]\s*/i; -const QUESTION_PROMPT_RE = - /^(?:what|which|when|where|why|how|who|whom|whose|do|does|did|can|could|would|will|should|is|are|am|may|shall)\b/i; -const USER_DIRECTED_PROMPT_RE = /\b(?:you|your|we|our)\b/i; -const USER_RESPONSE_CUE_RE = - /^(?:please\s+)?(?:confirm|reply|choose|pick|decide|advise)\b|^(?:please\s+)?answer\b|^(?:please\s+)?(?:let\s+me\s+know|tell\s+me)\b/i; - -function assistantText(message: AssistantMessage): string { - return message.content - .filter((content): content is TextContent => content.type === "text") - .map(content => content.text) - .join("\n") - .trim(); -} - -interface PromptLine { - text: string; - hadPromptLabel: boolean; -} - -function promptLine(line: string): PromptLine { - const withoutMarkdownPrefix = line.trim().replace(MARKDOWN_PROMPT_PREFIX_RE, "").trim(); - const withoutPromptLabel = withoutMarkdownPrefix.replace(PROMPT_LABEL_RE, "").trim(); - return { - text: withoutPromptLabel, - hadPromptLabel: withoutPromptLabel !== withoutMarkdownPrefix, - }; -} - -function isQuestionPromptLine(line: string): boolean { - const candidate = promptLine(line); - if (!/[??]\s*$/.test(candidate.text)) return false; - return ( - candidate.hadPromptLabel || - QUESTION_PROMPT_RE.test(candidate.text) || - USER_DIRECTED_PROMPT_RE.test(candidate.text) - ); -} - -function isResponseCueLine(line: string): boolean { - const candidate = promptLine(line) - .text.replace(/[.!?。!?]+$/, "") - .trim(); - return USER_RESPONSE_CUE_RE.test(candidate); -} - -function isAwaitingUserAnswer(message: AssistantMessage): boolean { - const text = assistantText(message); - if (!text) return false; - const lastLine = text.split(/\r?\n/).at(-1)?.trim(); - return lastLine !== undefined && (isQuestionPromptLine(lastLine) || isResponseCueLine(lastLine)); -} -/** `customType` for the hidden mid-run todo nudge; `display: false`, so it reaches - * the model but never renders in the TUI or transcript. */ -const MID_RUN_TODO_NUDGE_MESSAGE_TYPE = "mid-run-todo-nudge"; -/** Hidden plan nudge injected by prewalk; scrubbed from the LLM context - * when the switch happens. */ -const PREWALK_PLAN_MESSAGE_TYPE = "prewalk-plan"; -/** Hidden safety-net nudge forcing one more turn after a text-only reply to - * the plan nudge, which would otherwise end the run with no code written. */ -const PREWALK_CONTINUE_MESSAGE_TYPE = "prewalk-continue"; -/** Hidden "verify before finishing" checklist steered into the run at the - * switch, aimed at the fast model's specific failure patterns: partial - * multi-site fixes, unnecessarily broad rewrites, and reported-test-only - * verification. */ -const PREWALK_CHECKLIST_MESSAGE_TYPE = "prewalk-checklist"; -/** Hidden steered notice announcing a mid-session `xd://` mount/unmount delta - * (see {@link AgentSession.#notifyXdevMountDelta}). */ -const XDEV_MOUNT_NOTICE_MESSAGE_TYPE = "xdev-mount-notice"; -/** Tools whose first successful call triggers the switch — once the todo - * gate is open (see {@link AgentSession.#prewalkTodoSeen}). Bash is - * deliberately excluded: it doubles as exploration (ls/cat) and fired - * turn-1 switches in practice. `todo` is deliberately NOT a trigger: firing - * at the todo init handed the fast model 100% of the implementation with - * zero started work and measurably regressed pass rates. */ -const PREWALK_ACTION_TOOLS: Record = { - edit: true, - write: true, -}; -/** `customType` for the hidden hand-off message steered to the target model - * once PlanYolo auto-approves the plan. Unlike prewalk's plan nudge this - * is never scrubbed — it IS the instruction the target model acts on. */ -const PLAN_YOLO_HANDOFF_MESSAGE_TYPE = "plan-yolo-handoff"; -/** Abort reason for the Gemini reasoning-header runaway interrupt. Surfaced on the - * discarded assistant turn only; never reaches the model. */ -const GEMINI_HEADER_INTERRUPT_REASON = "Interrupted: emit a tool call instead of more planning"; -/** `customType` for the hidden tool-call reminder injected after the interrupt. */ -const GEMINI_TOOL_REMINDER_TYPE = "gemini-tool-call-reminder"; -/** `customType` for the hidden redirect notice injected into a turn retried after a - * thinking/response loop. Steers the model off the repeated content; never displayed. */ -const THINKING_LOOP_REDIRECT_TYPE = "thinking-loop-redirect"; -const TOOL_CALL_LOOP_REDIRECT_TYPE = "tool-call-loop-redirect"; - -function customMessageContentText(content: string | (TextContent | ImageContent)[]): string { - if (typeof content === "string") return content; - const parts: string[] = []; - for (const part of content) { - if (part.type === "text") parts.push(part.text); - } - return parts.join("\n"); -} - -function stringProperty(value: object, key: string): string | undefined { - const field = Object.getOwnPropertyDescriptor(value, key)?.value; - return typeof field === "string" ? field : undefined; -} - -function reportFromRewindReportContent(content: string): string { - const marker = "\nReport:\n"; - const index = content.lastIndexOf(marker); - const report = index >= 0 ? content.slice(index + marker.length) : content; - return report.trim(); -} - -type SemanticCheckpointToolName = "checkpoint" | "rewind"; - -type SemanticToolResult = { - toolName: SemanticCheckpointToolName; - details?: unknown; -}; - -/** - * Normalize checkpoint/rewind results across native calls and `write xd://` - * dispatches. Xdev keeps the wrapped tool's result details under `xdev.inner`, - * while direct calls put them on the result itself. - */ -function semanticToolResult(toolName: string | undefined, result: unknown): SemanticToolResult | undefined { - if (toolName === "checkpoint" || toolName === "rewind") { - const details = result && typeof result === "object" && "details" in result ? result.details : undefined; - return { toolName, details }; - } - const dispatch = writeDeviceDispatch(toolName ?? "", result); - if (dispatch?.mode !== "execute" || (dispatch.tool !== "checkpoint" && dispatch.tool !== "rewind")) { - return undefined; - } - return { toolName: dispatch.tool, details: dispatch.inner }; -} - -function isTodoPhase(value: unknown): value is TodoPhase { - if (!isRecord(value) || typeof value.name !== "string" || !Array.isArray(value.tasks)) return false; - return value.tasks.every( - task => - isRecord(task) && - typeof task.content === "string" && - (task.status === "pending" || - task.status === "in_progress" || - task.status === "completed" || - task.status === "abandoned"), - ); -} - -function completedRewindFromEntry(entry: SessionEntry): CompletedRewindState | undefined { - if (entry.type !== "custom_message" || entry.customType !== "rewind-report") return undefined; - const details = entry.details; - if (!details || typeof details !== "object") return undefined; - const startedAt = stringProperty(details, "startedAt"); - const rewoundAt = stringProperty(details, "rewoundAt"); - if (!startedAt || !rewoundAt) return undefined; - const report = - stringProperty(details, "report")?.trim() || - reportFromRewindReportContent(customMessageContentText(entry.content)); - return report.length > 0 ? { report, startedAt, rewoundAt } : undefined; -} -function isSuccessfulCheckpointEntry( - entry: SessionEntry, -): entry is SessionEntry & { type: "message"; message: Extract } { - if (entry.type !== "message" || entry.message.role !== "toolResult" || entry.message.isError === true) { - return false; - } - return semanticToolResult(entry.message.toolName, entry.message)?.toolName === "checkpoint"; -} - -function checkpointStartedAtFromEntry(entry: SessionEntry): string | undefined { - if (!isSuccessfulCheckpointEntry(entry)) return undefined; - const details = semanticToolResult(entry.message.toolName, entry.message)?.details; - if (details && typeof details === "object") { - const startedAt = stringProperty(details, "startedAt"); - if (startedAt) return startedAt; - } - return entry.timestamp; -} - -// A side-channel assistant response is signed for the hidden prompt/history that -// produced it. If we persist that response under a different user turn, native -// replay anchors become invalid; keep only visible, non-cryptographic content. -function sanitizeAssistantForReparentedHistory(message: AssistantMessage): AssistantMessage { - const content: AssistantMessage["content"] = []; - for (const block of message.content) { - if (block.type === "redactedThinking") continue; - if (block.type === "thinking") { - content.push({ type: "thinking", thinking: block.thinking }); - continue; - } - content.push(block); - } - return { ...message, content, providerPayload: undefined }; -} - -/** Session-specific events that extend the core AgentEvent */ -export type AgentSessionEvent = - | Exclude - | (Extract & { - /** False when an async delivery will resume the session before its true final settle. */ - isTerminal?: boolean; - }) - | { - type: "auto_compaction_start"; - reason: "threshold" | "overflow" | "idle" | "incomplete"; - action: "context-full" | "handoff" | "shake" | "snapcompact"; - } - | { - type: "auto_compaction_end"; - action: "context-full" | "handoff" | "shake" | "snapcompact"; - result: CompactionResult | undefined; - aborted: boolean; - willRetry: boolean; - errorMessage?: string; - /** True when compaction was skipped for a benign reason (no model, no candidates, nothing to compact). */ - skipped?: boolean; - } - | { - type: "auto_retry_start"; - attempt: number; - maxAttempts: number; - delayMs: number; - errorMessage: string; - errorId?: number; - } - | { - type: "auto_retry_end"; - success: boolean; - attempt: number; - finalError?: string; - recoveredErrors?: RecoveredRetryError[]; - } - | { type: "retry_fallback_applied"; from: string; to: string; role: string } - | { type: "retry_fallback_succeeded"; model: string; role: string } - | { type: "ttsr_triggered"; rules: Rule[] } - | { type: "todo_reminder"; todos: TodoItem[]; attempt: number; maxAttempts: number } - | { type: "todo_auto_clear" } - | { type: "irc_message"; message: CustomMessage } - | { type: "notice"; level: "info" | "warning" | "error"; message: string; source?: string } - | { - type: "thinking_level_changed"; - thinkingLevel: ThinkingLevel | undefined; - /** The user-configured selector when it differs from the effective level (e.g. `auto`). */ - configured?: ConfiguredThinkingLevel; - /** The level `auto` resolved to this turn, once classified. */ - resolved?: Effort; - } - | { type: "goal_updated"; goal: Goal | null; state?: GoalModeState }; -/** Listener function for agent session events */ -export type AgentSessionEventListener = (event: AgentSessionEvent) => void; - -const UNEXPECTED_STOP_MAX_RETRIES = 3; -const UNEXPECTED_STOP_TIMEOUT_MS = 4000; -const EMPTY_STOP_MAX_RETRIES = 3; -const RETRY_BACKOFF_MAX_DELAY_MS = 8_000; -/** - * Budget for callers on the user-visible `/quit` / `/exit` shutdown path that - * want to cap how long they wait for `MnemopiSessionState.dispose()` to finish - * its consolidate pass. Consolidate fires fresh LLM fact extractions, each a - * 1–3 s round-trip, so interactive shutdown passes this budget to keep the - * UI responsive. Callers that keep the process/session host alive must omit it - * so dispose still awaits the full consolidate-then-close pipeline. - */ -export const SHUTDOWN_CONSOLIDATE_BUDGET_MS = 1_500; - -export interface AgentSessionDisposeOptions { - mnemopiConsolidateTimeoutMs?: number; - /** - * Postmortem reason that triggered this dispose (signal/fatal teardown - * paths). When set, the persisted `session_exit` diagnostic records it - * instead of the generic `"dispose"` used for normal programmatic disposal - * (`/quit`, test teardown, subagent completion). - */ - reason?: postmortem.Reason; -} - -type CompactionCheckResult = Readonly<{ - deferredHandoff: boolean; - continuationScheduled: boolean; - automaticContinuationBlocked?: boolean; - historyRewritten?: boolean; -}>; - -const COMPACTION_CHECK_NONE: CompactionCheckResult = { - deferredHandoff: false, - continuationScheduled: false, -}; -const COMPACTION_CHECK_DEFERRED_HANDOFF: CompactionCheckResult = { - deferredHandoff: true, - continuationScheduled: false, -}; -const COMPACTION_CHECK_CONTINUATION: CompactionCheckResult = { - deferredHandoff: false, - continuationScheduled: true, -}; -const COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION: CompactionCheckResult = { - deferredHandoff: false, - continuationScheduled: false, - automaticContinuationBlocked: true, -}; - -/** - * User-facing notice for a compaction dead end: maintenance freed too little - * to retry safely. `remedies` names the recovery actions left on the emitting - * path — by the time the post-pass dead end fires, the tiered rescue has - * already attempted both elide and image-drop automatically. - */ -function compactionDeadEndWarning(remedies: string): string { - return ( - "Compaction freed too little context to make progress — pausing automatic maintenance to avoid a compaction loop. " + - `The most recent turn alone is too large to reduce further; ${remedies} or switch to a larger-context model.` - ); -} - -function createCodexCompactionContext(options: { - trigger: CodexCompactionContext["trigger"]; - reason: CodexCompactionContext["reason"]; - phase: CodexCompactionContext["phase"]; -}): CodexCompactionContext { - return { - operationId: crypto.randomUUID(), - trigger: options.trigger, - reason: options.reason, - phase: options.phase, - strategy: "memento", - }; -} - -/** - * Per-turn prune cache window. A tool result whose all-message suffix exceeds - * this is in the warm, already-sent prompt-cache prefix: re-writing it costs the - * cacheWrite premium on the whole suffix. Per-turn passes only reclaim inside - * this tail (matches the supersede pass's default `suffixTokenLimit`); deeper - * stale/age victims are left to compaction/shake, which rebuild the cache anyway. - */ -const PRUNE_CACHE_WARM_SUFFIX_TOKENS = 8_000; - -/** - * Idle gap after which the supersede pass may flush the whole sent region (the - * provider cache is cold, so re-writing it is free). MUST exceed the maximum - * Anthropic prompt-cache TTL — "long" retention (the OAuth default) is 1h — or a - * still-warm prefix is busted by the flush. 90 min leaves margin over the 1h TTL. - */ -const PRUNE_IDLE_FLUSH_MS = 90 * 60_000; -export type CommandMetadataChangedListener = () => void | Promise; -export type AsyncJobSnapshotItem = Pick; - -const RETRY_BACKOFF_JITTER_RATIO = 0.25; -/** - * Hysteresis band for the post-maintenance "did we actually create headroom?" - * check shared by the shake tail and the context-full / snapcompact tail. A - * pass counts as having resolved threshold pressure only when residual context - * lands at or below `COMPACTION_RECOVERY_BAND × threshold`. Re-checking against - * the raw threshold lets a pass keep reclaiming a trickle of the previous - * turn's output and land just under the line every turn, sustaining the - * auto-continue dead loop reported in #2275; the same band stops the - * context-full / snapcompact tail from re-firing on a history whose single - * most-recent kept turn already exceeds the threshold (the snapcompact thrash). - */ -const COMPACTION_RECOVERY_BAND = 0.8; - -function calculateRetryBackoffDelayMs(baseDelayMs: number, attempt: number): number { - const cappedDelayMs = Math.min(Math.max(0, baseDelayMs) * 2 ** Math.max(0, attempt - 1), RETRY_BACKOFF_MAX_DELAY_MS); - const jitter = 1 - Math.random() * RETRY_BACKOFF_JITTER_RATIO; - return cappedDelayMs * jitter; -} - -/** - * Slack added past a sibling credential's block expiry before retrying, so - * the next getApiKey lands after the block has actually lapsed. - */ -const SIBLING_UNBLOCK_BUFFER_MS = 1_000; -const NON_WHITESPACE_RE = /\S/; - -function hasNonWhitespace(value: string): boolean { - return NON_WHITESPACE_RE.test(value); -} - -export interface AsyncJobSnapshot { - running: AsyncJobSnapshotItem[]; - recent: AsyncJobSnapshotItem[]; - delivery: AsyncJobDeliveryState; -} - -export type { ShakeMode, ShakeResult }; -/** - * Prewalk: switches an active session one-way from its starting model to - * a fast/cheap `target` at the first completed turn that runs an edit/write - * tool once the todo list exists. A hidden plan nudge asks the starting - * model to write a plan, initialize its todo list from it, and start; the - * todo call opens the trigger gate (it never fires the switch itself), so - * the starting model always begins the implementation. A hidden - * checklist nudge asks the target model to verify its work before - * finishing. Both are always on — this is the one mechanism that won out - * over turn-count and ungated variants in testing. - */ -export interface Prewalk { - target: Model; - thinkingLevel?: ConfiguredThinkingLevel; -} - -/** - * PlanYolo: forces the session into read-only plan mode at start, then - * auto-approves the plan the instant the model calls `resolve({ action: - * "apply" })` for it — no interactive review — and switches to a fast/cheap - * `target` model to implement it. The headless counterpart to interactive - * plan mode's "Approve and execute", for print/non-interactive runs where - * there is no one to click Approve. - */ -export interface PlanYolo { - target: Model; - thinkingLevel?: ConfiguredThinkingLevel; -} - -// ============================================================================ -// Types -// ============================================================================ - -/** Identifies a retry fallback chain already entered during startup model resolution. */ -export interface InitialRetryFallbackState { - /** Role whose configured primary was unavailable. */ - role: string; - /** Configured primary selector retained for restoration when it becomes available. */ - originalSelector: string; - /** Thinking selector configured for the unavailable primary. */ - originalThinkingLevel: ConfiguredThinkingLevel | undefined; -} - -export interface AgentSessionConfig { - agent: Agent; - sessionManager: SessionManager; - settings: Settings; - /** Whether the caller explicitly requested yolo/auto-approve behavior for this session. */ - autoApprove?: boolean; - /** Models to cycle through with Ctrl+P (from --models flag) */ - scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; - /** Initial session thinking selector. */ - thinkingLevel?: ConfiguredThinkingLevel; - /** Retry chain ownership when startup selected one of its fallback entries. */ - initialRetryFallback?: InitialRetryFallbackState; - /** Prewalk from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ - prewalk?: Prewalk; - /** Force read-only plan mode at start, auto-approve on the model's first - * plan proposal (`write xd://propose`), then switch to the target to implement. */ - planYolo?: PlanYolo; - - /** Initial per-family service tiers (OpenAI / Anthropic / Google) for the live session. */ - serviceTierByFamily?: ServiceTierByFamily; - /** Prompt templates for expansion */ - promptTemplates?: PromptTemplate[]; - /** File-based slash commands for expansion */ - slashCommands?: FileSlashCommand[]; - /** Extension runner (created in main.ts with wrapped tools) */ - extensionRunner?: ExtensionRunner; - /** Loaded skills (already discovered by SDK) */ - skills?: Skill[]; - /** Skill loading warnings (already captured by SDK) */ - skillWarnings?: SkillWarning[]; - /** Whether runtime reloads may rediscover disk-backed skills for this session. */ - skillsReloadable?: boolean; - /** Custom commands (TypeScript slash commands) */ - customCommands?: LoadedCustomCommand[]; - skillsSettings?: SkillsSettings; - /** Agent directory used when applying memory backend changes during a live session. */ - memoryAgentDir?: string; - /** Recursion depth used to suppress live backend replacement in subagents. */ - memoryTaskDepth?: number; - /** Creates the built-in memory tools allowed by the current backend selection. */ - createMemoryTools?: () => Promise; - /** Model registry for API key resolution and model discovery */ - modelRegistry: ModelRegistry; - /** Tool registry for LSP and settings */ - toolRegistry?: Map; - /** Creates the tools registered only while `/vibe` mode is active. */ - createVibeTools?: () => AgentTool[]; - /** Tool names whose current registry entry is still the built-in implementation. */ - builtInToolNames?: Iterable; - /** Update tool-session predicates that render guidance from the live active tool set. */ - setActiveToolNames?: (names: Iterable) => void; - /** Register the write transport lazily when runtime xdev mounts first need it. */ - ensureWriteRegistered?: () => Promise; - /** Current session pre-LLM message transform pipeline */ - transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise; - /** - * Per-request transform applied after `convertToLlm` and before the - * provider call. Used for snapcompact, secret obfuscation, and image - * clamping. When supplied via {@link createAgentSession}, the advisor agent - * inherits this so its requests undergo the same shaping as the main turn. - */ - transformProviderContext?: (context: Context, model: Model) => Context | Promise; - /** - * Stream wrapper passed to side-channel requests (`/btw`, `/omfg`, IRC - * auto-replies, and handoff generation) so they apply the same provider - * shaping and host-level request wrappers as normal agent turns. Defaults - * to plain `streamSimple` when omitted. - */ - sideStreamFn?: StreamFn; - /** - * Stream wrapper passed to the advisor agent so its requests apply the - * session's `providers.openrouterVariant`, `providers.antigravityEndpoint`, - * `providers.maxInFlightRequests`, and `model.loopGuard.*` settings — - * keeping OpenRouter sticky-routing / response caching consistent with the - * main agent. Defaults to plain `streamSimple` when omitted. - */ - advisorStreamFn?: StreamFn; - /** Hint that OpenAI Codex requests should prefer websocket transport when supported. */ - preferWebsockets?: boolean; - /** Provider payload hook used by the active session request path */ - onPayload?: SimpleStreamOptions["onPayload"]; - /** Provider response hook used by the active session request path */ - onResponse?: SimpleStreamOptions["onResponse"]; - /** Raw SSE hook used by the active session request path */ - onSseEvent?: SimpleStreamOptions["onSseEvent"]; - /** Per-session raw SSE diagnostic buffer */ - rawSseDebugBuffer?: RawSseDebugBuffer; - /** Current session message-to-LLM conversion pipeline */ - convertToLlm?: (messages: AgentMessage[]) => Message[] | Promise; - /** System prompt builder that can consider tool availability. Returns ordered provider-facing blocks. */ - rebuildSystemPrompt?: (toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>; - /** Local calendar date provider used by prompt-cache invalidation. Defaults to the host local date. */ - getLocalCalendarDate?: () => string; - /** Entries of tools mounted under `xd://` (name + one-line summary), for /tools display. */ - getXdevToolEntries?: () => Array<{ name: string; summary: string }>; - /** Session-owned `xd://` device registry; reconciled as the active tool set changes. */ - xdevRegistry?: XdevRegistry; - /** Discoverable tools mounted under `xd://` in the initial enabled set (startup partition in `sdk.ts`). */ - initialMountedXdevToolNames?: string[]; - /** Explicit/effective names pinned top-level during runtime repartitioning. */ - presentationPinnedToolNames?: ReadonlySet; - /** - * Optional accessor for live MCP server instructions. Read by the session's - * `rebuildSystemPrompt`-skip optimization to detect server-side instruction - * changes (e.g. an MCP server upgrade) that would otherwise pass the tool-set - * signature comparison and silently keep a stale prompt cached. - */ - getMcpServerInstructions?: () => Map | undefined; - /** TTSR manager for time-traveling stream rules */ - ttsrManager?: TtsrManager; - /** Secret obfuscator for deobfuscating streaming edit content */ - obfuscator?: SecretObfuscator; - /** Inherited eval executor session id from a parent agent. */ - parentEvalSessionId?: string; - /** Logical owner for retained eval kernels created by this session. */ - evalKernelOwnerId?: string; - /** - * AsyncJobManager that this session installed as the process-global instance. - * Only set for top-level sessions; subagents inherit the parent's manager and - * **MUST NOT** dispose it on their own teardown. - */ - ownedAsyncJobManager?: AsyncJobManager; - /** - * AsyncJobManager reachable by this session for scoped job actions. - * - * Top-level owners receive their own manager, subagents receive the inherited - * parent manager, and secondary in-process top-level sessions receive - * `undefined` so job snapshots and ACP drains cannot observe the primary's - * state. - */ - asyncJobManager?: AsyncJobManager; - /** Agent identity (registry id like "Main" or "Alice") used for IRC routing. */ - agentId?: string; - /** Whether this session is the top-level agent or a subagent. Drives eager-task - * prelude gating so a top-level session created with a custom `agentId` still - * receives the always-mode reminder. Defaults to "main". */ - agentKind?: "main" | "sub"; - /** - * Override the provider-facing session ID for all API requests from this session. - * When absent, `sessionManager.getSessionId()` is used. Needed when benchmark or - * SDK callers issue probes / prewarming with an explicit `--provider-session-id` - * so that credential sticky selection is consistent with the session's streaming calls. - */ - providerSessionId?: string; - /** Marks `agent.promptCacheKey` as fork-inherited so incompatible route changes can clear it. */ - providerPromptCacheKeySource?: "explicit" | "fork"; - /** - * Full advisor toolset, pre-built in `createAgentSession` against a distinct, - * advisor-scoped `ToolSession` (its own `-advisor` session/agent id) so the - * advisor's tool state stays isolated from the primary. The advisor is a full - * agent; its config `tools` selects a subset (default read/grep/glob). Undefined - * when the advisor is disabled. - */ - advisorTools?: AgentTool[]; - /** Preloaded watchdog prompt content for the advisor. */ - advisorWatchdogPrompt?: string; - /** Preloaded YAML top-level `instructions` shared baseline, kept separate from - * `advisorWatchdogPrompt` so `/advisor configure` can swap it live. */ - advisorSharedInstructions?: string; - /** - * Preloaded project context files (AGENTS.md, etc.) rendered as a system-prompt - * block for the advisor — the same standing instructions the primary agent - * receives, so the reviewer holds the agent to them. - */ - advisorContextPrompt?: string; - /** - * Advisors discovered from `WATCHDOG.yml`. Empty/undefined runs a single - * legacy advisor on the `advisor` role (byte-for-byte the pre-config path). - */ - advisorConfigs?: AdvisorConfig[]; - /** - * Strip tool descriptions from provider-bound tool specs on side requests - * (handoff). Must match the session-start value used to build the system - * prompt so inline descriptors are not also sent through provider schemas. - */ - pruneToolDescriptions?: boolean; - /** - * Disconnect this session's OWNED MCP manager on dispose. Provided only when - * the session created the manager (top-level sessions); subagents reuse a - * parent's manager via `options.mcpManager` and omit this so a child's - * teardown never tears down the shared servers. - */ - disconnectOwnedMcpManager?: () => Promise; - /** - * Override the bundled system prompt used by automatic session-title - * generation paths (initial title + replan refresh). Source-of-truth is - * `TITLE_SYSTEM.md` discovered via {@link discoverTitleSystemPromptFile} and - * resolved through {@link resolvePromptInput}; refresh after a `/move`-style - * cwd change via {@link AgentSession.setTitleSystemPrompt}. - */ - titleSystemPrompt?: string; -} - -/** Options for AgentSession.prompt() */ -export interface PromptOptions { - /** Whether to expand file-based prompt templates (default: true) */ - expandPromptTemplates?: boolean; - /** Image attachments */ - images?: ImageContent[]; - /** When streaming, how to queue the message: "steer" (interrupt) or "followUp" (wait). */ - streamingBehavior?: "steer" | "followUp"; - /** Optional tool choice override for the next LLM call. */ - toolChoice?: ToolChoice; - /** Send as developer/system message instead of user. Providers that support it use the developer role; others fall back to user. */ - synthetic?: boolean; - /** Marks this prompt as a deliberate user action (typed message, `.`/`c` - * continue). Clears advisor auto-resume suppression that a user interrupt set. - * Defaults to `!synthetic`; manual-continue is synthetic yet user-initiated, so - * it sets this explicitly. Agent-initiated synthetic prompts (auto-continue, - * plan re-prime, reminders) leave it unset and keep suppression latched. */ - userInitiated?: boolean; - /** Explicit billing/initiator attribution for the prompt. Defaults to user prompts as `user` and synthetic prompts as `agent`. */ - attribution?: MessageAttribution; - /** Skip pre-send compaction checks for this prompt (internal use for maintenance flows). */ - skipCompactionCheck?: boolean; -} - -/** Options for AgentSession.followUp() */ -export interface FollowUpOptions { - /** Enqueue as a hidden developer message (agent-attributed by default) instead of a user follow-up. */ - synthetic?: boolean; - /** Whether to expand file-based prompt templates (default: true). */ - expandPromptTemplates?: boolean; - /** Explicit billing/initiator attribution. Defaults to `agent` for synthetic follow-ups. */ - attribution?: MessageAttribution; -} - -/** Result from a handoff operation. */ -export interface HandoffResult { - document: string; - savedPath?: string; -} - -export interface SessionHandoffOptions { - autoTriggered?: boolean; - signal?: AbortSignal; - onSwitchCancelled?: () => void; -} - -/** Result from cycleModel() */ -export interface ModelCycleResult { - model: Model; - thinkingLevel: ThinkingLevel | undefined; - /** Whether cycling through scoped models (--models flag) or all available */ - isScoped: boolean; -} - -/** Result from cycleRoleModels() */ -export interface RoleModelCycleResult { - model: Model; - thinkingLevel: ThinkingLevel | undefined; - role: string; -} - -/** A configured role resolved to a concrete model, used by role cycling and - * the plan-approval model slider. */ -export interface ResolvedRoleModel { - role: string; - model: Model; - thinkingLevel?: ConfiguredThinkingLevel; - explicitThinkingLevel: boolean; -} - -/** The set of resolvable role models plus the index of the currently active - * one within {@link ResolvedRoleModel.role} order. */ -export interface RoleModelCycle { - models: ResolvedRoleModel[]; - currentIndex: number; -} - -export interface ContextUsageBreakdown { - contextWindow: number; - anchored: boolean; - usedTokens: number; - systemPromptTokens: number; - systemToolsTokens: number; - systemContextTokens: number; - skillsTokens: number; - messagesTokens: number; -} - -/** Session statistics for /session command */ -export interface SessionStats { - sessionFile: string | undefined; - sessionId: string; - userMessages: number; - assistantMessages: number; - toolCalls: number; - toolResults: number; - totalMessages: number; - tokens: { - input: number; - output: number; - reasoning: number; - cacheRead: number; - cacheWrite: number; - total: number; - }; - premiumRequests: number; - cost: number; - contextUsage?: ContextUsage; -} - -/** Advisor statistics for /advisor status command. */ -export interface AdvisorStats { - configured: boolean; - active: boolean; - model?: Model; - contextWindow: number; - contextTokens: number; - tokens: { - input: number; - output: number; - reasoning: number; - cacheRead: number; - cacheWrite: number; - total: number; - }; - cost: number; - messages: { - user: number; - assistant: number; - total: number; - }; - /** Per-advisor breakdown; one entry per active advisor (single-advisor sessions have one). */ - advisors: PerAdvisorStat[]; -} - -/** One advisor's slice of {@link AdvisorStats}. Active advisors carry full - * token/cost data; disabled/no-model/quota-exhausted advisors appear with - * just `name` + `status` so the status line can render a dot for every - * configured advisor. */ -export interface PerAdvisorStat { - name: string; - status: AdvisorRuntimeStatus; - model?: Model; - contextWindow: number; - contextTokens: number; - tokens: AdvisorStats["tokens"]; - cost: number; - messages: AdvisorStats["messages"]; - sessionId?: string; -} - -/** - * One live advisor instance: its own agent/runtime/tools/recorder plus a - * per-advisor emission guard and identity. The session holds an array of these; - * primary-scoped state (turn counters, interrupt latches, the shared yield - * channel) stays on the session. - */ -interface AdvisorRetryFallbackState { - role: string; - originalSelector: string; - originalThinkingLevel: ThinkingLevel; - lastAppliedThinkingLevel: ThinkingLevel; -} - -interface ActiveAdvisor { - /** Display name from config ("default" for the legacy no-YAML advisor). */ - name: string; - /** Slug for the transcript filename/session id; "" → `__advisor.jsonl`. */ - slug: string; - agent: Agent; - runtime: AdvisorRuntime; - adviseTool: AdviseTool; - emissionGuard: AdvisorEmissionGuard; - recorder: AdvisorTranscriptRecorder; - /** Latest recorder close, awaited by dispose() so the final turn lands on disk. */ - recorderClosed: Promise; - /** Unsubscribe for the advisor agent's event stream feeding the recorder. */ - agentUnsubscribe?: () => void; - model: Model; - thinkingLevel: ThinkingLevel; - /** Provider credential/session identity retained across advisor model switches. */ - providerSessionId: string | undefined; - /** Active chain state retained until the configured primary can be restored. */ - retryFallback?: AdvisorRetryFallbackState; - /** A switched advisor model has not yet completed its first successful turn. */ - retryFallbackPendingSuccess: boolean; - /** Stable key for the resolved runtime inputs that require a rebuild to change. */ - signature: string; -} - -/** Runtime-only advisor compaction metadata. It never enters the model-facing summary text. */ -interface AdvisorCompactionSummaryMessage extends CompactionSummaryMessage { - firstKeptEntryId?: string; - /** First message index eligible to anchor provider usage after this compaction. */ - advisorUsageAnchorStartIndex?: number; -} - -/** Resolved advisor config ready to instantiate as an {@link ActiveAdvisor}. */ -interface AdvisorRuntimeDescriptor { - config: AdvisorConfig; - name: string; - slug: string; - model: Model; - thinkingLevel: ThinkingLevel; - signature: string; -} - -export interface FreshSessionResult { - previousSessionId: string; - sessionId: string; - closedProviderSessions: number; -} - /** Internal marker for hook messages queued through the agent loop */ // ============================================================================ // Constants @@ -1316,129 +327,6 @@ export interface FreshSessionResult { /** Standard thinking levels */ -/** `retry.fallbackChains` config: chain key (role name or model selector) → ordered fallback selectors. */ -type RetryFallbackChains = Record; - -type RetryFallbackRevertPolicy = "never" | "cooldown-expiry"; - -interface RetryFallbackSelector { - raw: string; - provider: string; - id: string; - thinkingLevel: ThinkingLevel | undefined; -} - -interface ActiveRetryFallbackState { - /** Chain key that produced this fallback: a model-role name or a model-selector key. */ - role: string; - originalSelector: string; - originalThinkingLevel: ConfiguredThinkingLevel | undefined; - lastAppliedFallbackThinkingLevel: ConfiguredThinkingLevel | undefined; - pinned: boolean; -} - -function parseRetryFallbackSelector( - selector: string, - modelLookup?: { find(provider: string, id: string): Model | undefined }, -): RetryFallbackSelector | undefined { - const trimmed = selector.trim(); - if (!trimmed) return undefined; - const parsed = parseModelString(trimmed, { - allowMaxSuffix: true, - allowAutoAlias: true, - isLiteralModelId: (provider, id) => modelLookup?.find(provider, id) !== undefined, - }); - if (!parsed) return undefined; - return { - raw: trimmed, - provider: parsed.provider, - id: parsed.id, - thinkingLevel: concreteThinkingLevel(parsed.thinkingLevel), - }; -} - -/** - * `retry.fallbackChains` keys are either model-role names (`smol`, `default`) - * or model selectors (`provider/model-id[:thinking]`). Role names never - * contain a slash, so its presence marks a model-keyed chain whose primary is - * the key itself — the chain follows the model across role reassignments. - */ -function isRetryFallbackModelKey(key: string): boolean { - return key.includes("/"); -} - -/** - * A wildcard fallback-chain key/entry: `provider/*` matches any model of that - * provider; an id-prefixed `provider/prefix/*` (e.g. `openrouter/google/*`) - * scopes it to ids under that prefix — aggregators namespace model ids by - * upstream vendor. - */ -function isRetryFallbackWildcardKey(key: string): boolean { - return key.endsWith("/*"); -} - -/** - * Split a `…/*` wildcard key/entry into its provider and optional id prefix - * (`google-vertex/*` → provider only; `openrouter/google/*` → provider - * `openrouter`, prefix `google`). A template that names a known provider in - * full wins over the split, so provider ids containing `/` keep working. - */ -function parseRetryFallbackWildcard( - key: string, - isKnownProvider: (provider: string) => boolean, -): { provider: string; idPrefix: string | undefined } { - const template = key.slice(0, -2); - const slash = template.indexOf("/"); - if (slash < 0 || isKnownProvider(template)) return { provider: template, idPrefix: undefined }; - return { provider: template.slice(0, slash), idPrefix: template.slice(slash + 1) }; -} - -function formatRetryFallbackSelector(model: Model, thinkingLevel: ThinkingLevel | undefined): string { - return formatModelSelectorValue(formatModelStringWithRouting(model), thinkingLevel); -} - -function formatRetryFallbackBaseSelector(selector: RetryFallbackSelector): string { - return `${selector.provider}/${selector.id}`; -} - -const EPHEMERAL_REPLY_MAX_BYTES = 4096; - -/** - * Collapse degenerate ephemeral replies (/btw, /omfg side-channel turns). - * Models occasionally loop on a single line (~16 reports of N-times-repeated - * replies); compress runs longer than 3 down to one instance + `[…N×]`, then - * cap at 4 KiB so a runaway reply can't flood the channel. - */ -function dedupeEphemeralReply(text: string): string { - if (!text) return text; - const lines = text.split("\n"); - const out: string[] = []; - let i = 0; - while (i < lines.length) { - let j = i + 1; - while (j < lines.length && lines[j] === lines[i]) j++; - const runLen = j - i; - if (runLen > 3) { - out.push(lines[i], `[…${runLen}×]`); - } else { - for (let k = 0; k < runLen; k++) out.push(lines[i]); - } - i = j; - } - let result = out.join("\n"); - if (Buffer.byteLength(result, "utf8") > EPHEMERAL_REPLY_MAX_BYTES) { - // Trim by characters until we're under the byte budget — handles multi-byte - // glyphs at the boundary without splitting them. - const suffix = "\n[…truncated]"; - const budget = EPHEMERAL_REPLY_MAX_BYTES - Buffer.byteLength(suffix, "utf8"); - while (Buffer.byteLength(result, "utf8") > budget) { - result = result.slice(0, -1); - } - result += suffix; - } - return result; -} - /** * Build the per-request `metadata` payload for the Anthropic provider, shaped * like real Claude Code's `getAPIMetadata` output (`{ session_id, account_uuid, @@ -1513,323 +401,15 @@ const noOpUIContext: ExtensionUIContext = { setToolsExpanded: () => {}, }; -function createHandoffContext(document: string): string { - return `\n${document}\n\n\nThe above is a handoff document from a previous session. Use this context to continue the work seamlessly.`; -} - -function createHandoffFileName(date = new Date()): string { - const fileTimestamp = date.toISOString().replace(/[:.]/g, "-"); - return `handoff-${fileTimestamp}.md`; -} - -// ============================================================================ -// ACP Permission Gate -// ============================================================================ - -/** Tools that require user permission before execution when an ACP client is connected. */ -const PERMISSION_REQUIRED_TOOLS = new Set(["bash", "edit", "delete", "move"]); - -/** Permission options presented to the client on each gated tool call. */ -const PERMISSION_OPTIONS: ClientBridgePermissionOption[] = [ - { optionId: "allow_once", name: "Allow once", kind: "allow_once" }, - { optionId: "allow_always", name: "Always allow", kind: "allow_always" }, - { optionId: "reject_once", name: "Reject", kind: "reject_once" }, - { optionId: "reject_always", name: "Always reject", kind: "reject_always" }, -]; - -const PERMISSION_OPTIONS_BY_ID = new Map(PERMISSION_OPTIONS.map(option => [option.optionId, option])); - -function getStringProperty(value: Record, key: string): string | undefined { - const candidate = value[key]; - return typeof candidate === "string" ? candidate : undefined; -} - -function collectStringPaths(value: unknown): string[] { - return Array.isArray(value) ? value.filter((item): item is string => typeof item === "string") : []; -} - -function getEditDestructiveIntent(args: unknown): { kind: "delete" | "move"; paths: string[] } | undefined { - if (!args || typeof args !== "object" || Array.isArray(args)) return undefined; - const a = args as Record; - - const edits = Array.isArray(a.edits) ? a.edits : undefined; - if (edits) { - const path = getStringProperty(a, "path"); - if (path) { - for (const edit of edits) { - if (!edit || typeof edit !== "object" || Array.isArray(edit)) continue; - const op = getStringProperty(edit as Record, "op"); - if (op === "delete") return { kind: "delete", paths: [path] }; - } - } - for (const edit of edits) { - if (!edit || typeof edit !== "object" || Array.isArray(edit)) continue; - const entry = edit as Record; - const op = getStringProperty(entry, "op"); - const rename = getStringProperty(entry, "rename"); - if (op !== "create" && rename) return { kind: "move", paths: path ? [path, rename] : [rename] }; - } - } - - const input = getStringProperty(a, "input"); - if (input) { - try { - const patch = Patch.parse(input); - for (const section of patch.sections) { - if (section.fileOp?.kind === "rem") return { kind: "delete", paths: [section.path] }; - if (section.fileOp?.kind === "move") return { kind: "move", paths: [section.path, section.fileOp.dest] }; - } - } catch { - // Not a hashline patch — fall through to apply_patch parsing. - } - try { - const entries = expandApplyPatchToEntries({ input }); - const deleteEntry = entries.find(entry => entry.op === "delete"); - if (deleteEntry) return { kind: "delete", paths: [deleteEntry.path] }; - const moveEntry = entries.find(entry => entry.rename); - if (moveEntry?.rename) return { kind: "move", paths: [moveEntry.path, moveEntry.rename] }; - } catch { - // If the edit input is not an apply_patch envelope, it is not a delete/move operation. - } - } - - return undefined; -} - -function getPermissionIntent( - toolName: string, - args: unknown, -): { toolName: string; title: string; paths?: string[]; cacheKey: string } | undefined { - const a = args && typeof args === "object" && !Array.isArray(args) ? (args as Record) : {}; - if (toolName === "bash") { - const cmd = getStringProperty(a, "command")?.slice(0, 80); - return { toolName, title: cmd || toolName, cacheKey: toolName }; - } - if (toolName === "delete") { - const p = getStringProperty(a, "path"); - return { toolName, title: p ? `Delete ${p}` : toolName, paths: p ? [p] : undefined, cacheKey: toolName }; - } - if (toolName === "move") { - const from = getStringProperty(a, "oldPath") ?? getStringProperty(a, "path") ?? getStringProperty(a, "from"); - const to = getStringProperty(a, "newPath") ?? getStringProperty(a, "to") ?? getStringProperty(a, "destination"); - if (from && to) return { toolName, title: `Move ${from} to ${to}`, paths: [from, to], cacheKey: toolName }; - return { - toolName, - title: from ? `Move ${from}` : toolName, - paths: from ? [from] : undefined, - cacheKey: toolName, - }; - } - if (toolName === "edit") { - const intent = getEditDestructiveIntent(args); - if (!intent) return undefined; - if (intent.kind === "delete") { - return { - toolName, - title: `Delete ${intent.paths[0] ?? "edit target"}`, - paths: intent.paths, - cacheKey: "edit:delete", - }; - } - const from = intent.paths[0]; - const to = intent.paths[1]; - return { - toolName, - title: from && to ? `Move ${from} to ${to}` : `Move ${from ?? to ?? "edit target"}`, - paths: intent.paths, - cacheKey: "edit:move", - }; - } - return undefined; -} - -function extractPermissionLocations( - args: unknown, - cwd: string, - explicitPaths?: string[], -): { path: string; line?: number }[] { - if (!args || typeof args !== "object") return []; - const a = args as Record; - const out: { path: string; line?: number }[] = []; - const pushPath = (value: unknown) => { - if (typeof value !== "string" || value.length === 0) return; - // ACP locations carry file paths that the editor host will open or focus; - // they must be absolute or the client cannot resolve them. Resolve raw - // tool args (often cwd-relative) against the session cwd before sending. - let resolved: string; - try { - resolved = resolveToCwd(value, cwd); - } catch { - return; - } - if (out.some(location => location.path === resolved)) return; - out.push({ path: resolved }); - }; - if (explicitPaths) { - for (const p of explicitPaths) { - pushPath(p); - } - return out; - } - pushPath(a.path); - pushPath(a.file); - for (const p of collectStringPaths(a.paths)) { - pushPath(p); - } - pushPath(a.oldPath); - pushPath(a.newPath); - pushPath(a.from); - pushPath(a.to); - pushPath(a.source); - pushPath(a.destination); - return out; -} - // ============================================================================ // AgentSession Class // ============================================================================ -/** Entry returned by {@link AgentSession.clearQueue} / {@link AgentSession.popLastQueuedMessage}. */ -export type RestoredQueuedMessage = { text: string; images?: ImageContent[] }; - -function queuedTextContent(message: AgentMessage): string | undefined { - if (!("content" in message)) return undefined; - const content = message.content; - if (typeof content === "string") return content; - for (const part of content) { - if (part.type === "text") return part.text; - } - return undefined; -} - -function queuedImageContent(message: AgentMessage): ImageContent[] | undefined { - if (!("content" in message) || typeof message.content === "string") return undefined; - const images: ImageContent[] = []; - for (const part of message.content) { - if (part.type === "image" && typeof part.data === "string" && typeof part.mimeType === "string") { - images.push(part); - } - } - return images.length > 0 ? images : undefined; -} - -function isDisplayableQueuedMessage(message: AgentMessage): boolean { - return !(message.role === "custom" && message.display === false); -} - -function isAdvisorCard(message: AgentMessage): message is CustomMessage { - return message.role === "custom" && message.customType === "advisor"; -} - -/** - * Emit a warn-level log for a turn that ended in a provider error so recurring - * stream failures are diagnosable from the main log alone. The `agent_end` - * routing trace is debug-only and omits the error fields; without this a - * session dying on provider errors leaves only `stopReason:"error"` debug lines - * and the real cause lives solely in the session transcript (issue #6177). - * No-op for any non-error stop reason. - */ -export function logProviderTurnError(msg: AssistantMessage): void { - if (msg.stopReason !== "error") return; - logger.warn("agent turn ended with provider error", { - provider: msg.provider, - model: msg.model, - errorMessage: msg.errorMessage, - errorStatus: msg.errorStatus, - errorId: msg.errorId, - }); -} - -function isTerminalTextAssistantAnswer(message: AgentMessage | undefined): message is AssistantMessage { - if (message?.role !== "assistant" || message.stopReason !== "stop") return false; - let hasText = false; - for (const part of message.content) { - if (part.type === "toolCall") return false; - if (part.type === "text") { - if (part.text.trim().length > 0) hasText = true; - continue; - } - if (part.type === "thinking" || part.type === "redactedThinking" || part.type === "fallback") continue; - return false; - } - return hasText; -} - -/** - * A queued message the user can restore to the editor / pull back as a draft. - * Only genuinely user-authored messages qualify: plain user turns, or custom - * messages explicitly attributed to the user (e.g. `/skill` invocations). - * Agent-authored queued cards — advisor concern/blocker notes, IRC asides, - * extension notices, hidden goal/plan/budget steers — ride the same - * steer/follow-up queues but must never be dumped into the editor on Esc/Alt+Up. - */ -function isUserQueuedMessage(message: AgentMessage): boolean { - if (message.role === "user") return true; - return message.role === "custom" && message.attribution === "user" && message.display !== false; -} - -/** Custom-message types of the hidden magic-keyword notices that `#createMagicKeywordNotices` - * enqueues alongside a user prompt. Keep in sync with that method. */ -const MAGIC_KEYWORD_NOTICE_TYPES: ReadonlySet = new Set([ - "ultrathink-notice", - "orchestrate-notice", - "workflow-notice", -]); - -/** Custom-message type of the hidden companion carrying vision descriptions of image - * attachments sent to a text-only model (see `#buildImageDescriptionNotice`). */ -const IMAGE_ATTACHMENT_DESCRIPTION_TYPE = "image-attachment-description"; - -/** - * A hidden, user-attributed companion of a queued user prompt: the magic-keyword - * notices (`ultrathink`/`orchestrate`/`workflow`) enqueued alongside the user - * message. They are `attribution: "user"` but `display: false`, so they are not - * editor-restorable; when the user pulls their prompt back out of the queue these - * must leave with it rather than linger as stale, companion-less steering. Scoped to - * the known notice types so an unrelated hidden user custom is never silently dropped. - */ -function isHiddenUserCompanion(message: AgentMessage): boolean { - return ( - message.role === "custom" && - message.attribution === "user" && - message.display === false && - (MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType) || message.customType === IMAGE_ATTACHMENT_DESCRIPTION_TYPE) - ); -} - -function queueChipText(message: AgentMessage): string { - if (message.role === "custom") { - return readQueueChipText(message.details) ?? queuedTextContent(message) ?? ""; - } - const text = queuedTextContent(message) ?? ""; - if (text) return text; - return queuedImageContent(message) ? "[Image]" : ""; -} - -function toRestoredQueuedMessage(message: AgentMessage): RestoredQueuedMessage { - return { text: queueChipText(message), images: queuedImageContent(message) }; -} - -function mergeLlmCompactionPreserveData( - hookPreserveData: Record | undefined, - resultPreserveData: Record | undefined, -): Record | undefined { - const preserveData = { ...(hookPreserveData ?? {}), ...(resultPreserveData ?? {}) }; - return snapcompact.stripPreservedArchive(Object.keys(preserveData).length > 0 ? preserveData : undefined); -} - type MessageEndPersistenceSlot = { readonly promise: Promise; persist: (persistMessage: () => void) => Promise; release: () => void; }; -type PendingRecoveredRetryError = { - entryId: string; - persistenceKey: string; - recovery: AssistantRetryRecoveryKind; - attempt: number; - note: string; -}; type PostPromptSkipReason = "aborted" | "stale-generation"; @@ -1847,8 +427,6 @@ type ScheduledAgentContinueOptions = { onError?: () => void; }; -const REPLAN_TITLE_CONTEXT_TURN_LIMIT = 6; - type SessionTitleSource = "auto" | "user"; type SessionNameTrigger = "replan"; type SetSessionNameWithTrigger = ( @@ -1857,66 +435,6 @@ type SetSessionNameWithTrigger = ( trigger?: SessionNameTrigger, ) => Promise; -function isRecord(value: unknown): value is Record { - return value !== null && typeof value === "object" && !Array.isArray(value); -} - -function textFromContent(content: unknown): string { - if (typeof content === "string") return content.trim(); - if (!Array.isArray(content)) return ""; - const parts: string[] = []; - for (const block of content) { - if (!isRecord(block) || block.type !== "text" || typeof block.text !== "string") continue; - const text = block.text.trim(); - if (text) parts.push(text); - } - return parts.join("\n\n"); -} - -function thinkingFromContent(content: unknown): string { - if (!Array.isArray(content)) return ""; - const parts: string[] = []; - for (const block of content) { - if (!isRecord(block) || block.type !== "thinking" || typeof block.thinking !== "string") continue; - const thinking = block.thinking.trim(); - if (thinking) parts.push(thinking); - } - return parts.join("\n\n"); -} - -function toolCallOpFromMessage(message: AgentMessage, toolCallId: string): string | undefined { - if (message.role !== "assistant" || !Array.isArray(message.content)) return undefined; - for (const block of message.content) { - if (!isRecord(block) || block.type !== "toolCall" || block.id !== toolCallId) continue; - return isRecord(block.arguments) ? getStringProperty(block.arguments, "op") : undefined; - } - return undefined; -} - -function titleConversationTurnFromMessage(message: AgentMessage): TitleConversationTurn | undefined { - if (message.role !== "user" && message.role !== "assistant") return undefined; - const text = textFromContent(message.content); - const thinking = message.role === "assistant" ? thinkingFromContent(message.content) : undefined; - if (!text && !thinking) return undefined; - return { role: message.role, ...(text ? { text } : {}), ...(thinking ? { thinking } : {}) }; -} - -function syntheticToolResultTailStart(messages: readonly AgentMessage[]): number { - let index = messages.length; - while (index > 0 && isSyntheticToolResultMessage(messages[index - 1])) { - index--; - } - return index; -} - -function retryableAssistantTurnEnd(messages: readonly AgentMessage[]): number | undefined { - const turnEnd = syntheticToolResultTailStart(messages); - const message = messages[turnEnd - 1]; - if (message?.role !== "assistant") return undefined; - if (message.stopReason !== "error" && message.stopReason !== "aborted") return undefined; - return turnEnd; -} - export class AgentSession { readonly agent: Agent; readonly sessionManager: SessionManager; @@ -1925,32 +443,16 @@ export class AgentSession { getXdevToolEntries: () => Array<{ name: string; summary: string }>; readonly yieldQueue: YieldQueue; fileSnapshotStore?: InMemorySnapshotStore; - #autoApprove: boolean; #powerAssertion: MacOSPowerAssertion | undefined; readonly configWarnings: string[] = []; - #scopedModels: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; - /** Effective, metadata-clamped thinking level applied to the agent (never `auto`). */ - #thinkingLevel: ThinkingLevel | undefined; - /** True when the user configured `auto`; the effective level is resolved per turn. */ - #autoThinking: boolean = false; - /** The level `auto` last resolved to (for UI); undefined until a turn is classified. */ - #autoResolvedLevel: Effort | undefined; - #prewalk: Prewalk | undefined; - /** True once the plan nudge has been queued; scrubbed from context at the switch. */ - #prewalkPlanInjected = false; - /** Armed by plan/tool progress; consumed by one text-only continuation. */ - #prewalkContinuePending = false; - /** True once any successful `todo` call landed — opens the prewalk - * trigger gate: the switch fires at the first edit/write AFTER the todo - * list exists (sessions without an ACTIVE todo tool skip the gate). */ - #prewalkTodoSeen = false; - #planYolo: PlanYolo | undefined; - #planYoloPreviousTools: string[] | undefined; - #planYoloArmed = false; + readonly #models: ModelControls; + readonly #tools: SessionTools; + readonly #prewalk: PrewalkCoordinator; + readonly #providerBoundary: SessionProviderBoundary; #promptTemplates: PromptTemplate[]; #slashCommands: FileSlashCommand[]; @@ -1969,91 +471,31 @@ export class AgentSession { #pendingNextTurnMessages: CustomMessage[] = []; #scheduledHiddenNextTurnGeneration: number | undefined = undefined; #queuedMessageDrainScheduled = false; - /** Latched true when the user deliberately interrupts (USER_INTERRUPT_LABEL); - * suppresses advisor concern/blocker auto-resume until the user next resumes. - * Advisor advice is still recorded into the transcript, just not auto-run. */ - #advisorAutoResumeSuppressed = false; - /** Print-mode sessions preserve advisor notes without starting hidden primary turns. */ - #preserveAdvisorAdvice = false; - #advisorPrimaryTurnsCompleted = 0; - #advisorInterruptImmuneTurnStart: number | undefined; #planModeState: PlanModeState | undefined; #vibeModeState: VibeModeState | undefined; #goalModeState: GoalModeState | undefined; #goalRuntime: GoalRuntime; - #advisorEnabled = false; - #advisorTools?: AgentTool[]; - #advisorWatchdogPrompt?: string; - #advisorSharedInstructions?: string; - #advisorContextPrompt?: string; - #advisorYieldQueueUnsubscribe?: () => void; - /** Live advisors. Empty when no advisor is active. */ - #advisors: ActiveAdvisor[] = []; - /** Configured advisor roster from WATCHDOG.yml; undefined/empty → single legacy advisor. */ - #advisorConfigs?: AdvisorConfig[]; - /** Per-advisor runtime status (slug → {name, status}). Tracks disabled/quota/states - * for the configured roster even when the advisor has no live runtime. The name - * is stored alongside the status so {@link getAdvisorStats} doesn't need to - * recompute slugs or resolve config names. */ - #advisorStatuses: Map = new Map(); - /** Provider-facing UUIDv7 identities keyed by primary provider session and advisor slug. */ - #advisorProviderSessionIds = new Map(); - /** Aggregate of the most recent stop's recorder closes; awaited by dispose() and - * used as the open barrier for the next build so two writers never share a file. */ - #advisorRecorderClosed: Promise = Promise.resolve(); + readonly #advisors: SessionAdvisors; #goalTurnCounter = 0; #planReferenceSent = false; #planReferencePath = "local://PLAN.md"; #clientBridge: ClientBridge | undefined; #allowAcpAgentInitiatedTurns = false; - /** Per-session memory of allow_always / reject_always decisions for gated tools. */ - #acpPermissionDecisions: Map = new Map(); /** Session file created by this session's `/move`; removed on dispose if it stayed empty. */ #movedFromEmptySessionFile?: string; - // Compaction state - #compactionAbortController: AbortController | undefined = undefined; - #autoCompactionAbortController: AbortController | undefined = undefined; + readonly #maintenance: SessionMaintenance; // Branch summarization state #branchSummaryAbortController: AbortController | undefined = undefined; - // Handoff state - #handoffAbortController: AbortController | undefined = undefined; - #skipPostTurnMaintenanceAssistantTimestamp: number | undefined = undefined; + readonly #handoff: SessionHandoff; // Retry state - #retryAbortController: AbortController | undefined = undefined; - #retryAttempt = 0; - #retryPromise: Promise | undefined = undefined; - #retryResolve: (() => void) | undefined = undefined; - #activeRetryFallback: ActiveRetryFallbackState | undefined = undefined; - #pendingRecoveredRetryErrors: PendingRecoveredRetryError[] = []; - // Todo completion reminder state - #todoReminderCount = 0; - /** - * Set true after a todo reminder is appended; cleared when the agent makes any tool-level - * progress (toolResult) or a new user prompt arrives. Suppresses follow-up reminders within - * the same agent self-continuation chain so a text-only acknowledgement ("paused at your - * instruction") does not drive 1/3 → 2/3 → 3/3 without user input. - */ - #todoReminderAwaitingProgress = false; - /** - * Successful mutating tool results (bash/eval/edit/write/ast_edit) since the - * agent last touched the `todo` tool. Drives {@link #takeMidRunTodoNudge} so - * the live HUD stays in sync with actual progress instead of flipping - * `0/N -> N/N` only at the very end of a long run (issue #3651). Read-only - * tools and errored results never tick it. Reset to 0 on any `todo` tool - * result, on a nudge fire (cooldown), on a stop-time reminder, and at every - * new-prompt / clear / handoff lifecycle boundary. - */ - #mutationsSinceLastTodoTouch = 0; - /** Mid-run nudges fired this prompt cycle; capped by - * {@link MID_RUN_TODO_NUDGE_MAX_PER_CYCLE}, reset with the counter above. */ - #midRunNudgeCount = 0; + readonly #recovery: TurnRecovery; #planModeReminderCount = 0; #planModeReminderAwaitingProgress = false; - #todoPhases: TodoPhase[] = []; + readonly #todo: TodoTracker; #replanTitleRefreshInFlight: Promise | undefined = undefined; /** Resolved TITLE_SYSTEM.md override applied to every automatic session-title * generation path. Refresh via {@link AgentSession.setTitleSystemPrompt} when @@ -2062,15 +504,9 @@ export class AgentSession { #titleGenerationAbortController = new AbortController(); #toolChoiceQueue = new ToolChoiceQueue(); - // Bash execution state - #bashAbortControllers = new Set(); - #pendingBashMessages: PendingBashMessage[] = []; - #bashSessionTarget!: BashSessionTarget; + readonly #bash: BashRunner; - // Python execution state - #evalAbortControllers = new Set(); - #evalKernelOwnerId: string; - #parentEvalSessionId: string | undefined; + readonly #eval: EvalRunner; /** * AsyncJobManager owned by this session (top-level only). Subagents leave * this undefined and **MUST NOT** dispose the global instance on teardown. @@ -2086,15 +522,8 @@ export class AgentSession { readonly #asyncJobManager: AsyncJobManager | undefined; /** Clears this session's owner delivery sink registration; set when a manager + agent id exist. */ #unregisterAsyncDeliverySink: (() => void) | undefined; - #pendingPythonMessages: PythonExecutionMessage[] = []; - #activeEvalExecutions = new Set>(); - #evalExecutionDisposing = false; - // Incoming IRC messages received while a turn was streaming. Parent IRCs - // enter the steering queue; peer IRCs enter the interrupt queue and drain as - // asides at the next boundary; passive IRC records stay in the aside queue. - #pendingIrcInterrupts: CustomMessage[] = []; - #pendingIrcAsides: CustomMessage[] = []; + readonly #irc: IrcBridge; // Agent identity (registry id) used for IRC routing and job ownership. #agentId: string | undefined; #agentKind: "main" | "sub" = "main"; @@ -2115,84 +544,27 @@ export class AgentSession { #turnIndex = 0; #messageEndPersistenceTail: Promise = Promise.resolve(); #pendingMessageEndPersistence = new Map>(); - /** Async lifecycle handlers for visible advisor cards emitted outside the primary loop. */ - #pendingAdvisorCardEvents = new Set>(); #persistedMessageKeys: { anchor: string; keys: Set } | undefined; - #skills: Skill[]; - #skillWarnings: SkillWarning[]; - // Custom commands (TypeScript slash commands) #customCommands: LoadedCustomCommand[] = []; /** MCP prompt commands (updated dynamically when prompts are loaded) */ #mcpPromptCommands: LoadedCustomCommand[] = []; - #skillsSettings: SkillsSettings | undefined; - #skillsReloadable: boolean; - // Model registry for API key resolution #modelRegistry: ModelRegistry; - // Tool registry and prompt builder for extensions - #toolRegistry: Map; - #createVibeTools: (() => AgentTool[]) | undefined; - #installedVibeToolNames = new Set(); #transformContext: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise; #onPayload: SimpleStreamOptions["onPayload"] | undefined; #onResponse: SimpleStreamOptions["onResponse"] | undefined; #onSseEvent: SimpleStreamOptions["onSseEvent"] | undefined; - #transformProviderContext: ((context: Context, model: Model) => Context | Promise) | undefined; #sideStreamFn: StreamFn; - #advisorStreamFn: StreamFn | undefined; #preferWebsockets: boolean | undefined; #convertToLlm: (messages: AgentMessage[]) => Message[] | Promise; - #rebuildSystemPrompt: - | ((toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>) - | undefined; - #getLocalCalendarDate: () => string; - #getMcpServerInstructions: (() => Map | undefined) | undefined; - #setActiveToolNames: ((names: Iterable) => void) | undefined; - #ensureWriteRegistered: (() => Promise) | undefined; #disconnectOwnedMcpManager: (() => Promise) | undefined; - #presentationPinnedToolNames: ReadonlySet | undefined; - #runtimeSelectedToolNames: ReadonlySet | undefined; - #baseSystemPrompt: string[]; - #baseSystemPromptBeforeMemoryPromotion: string[] | undefined; - /** - * Signature of the (toolNames, tool descriptions) tuple passed to the most - * recent successful `rebuildSystemPrompt` call. Used to skip redundant rebuilds - * when MCP servers reconnect without changing their tool definitions, which is - * the dominant cause of prompt-cache invalidation in long sessions. - */ - #lastAppliedToolSignature: string | undefined; - /** - * Model identifier (`provider/id`) currently rendered into `#baseSystemPrompt`. - * The prompt surfaces the active model to the agent, so a model switch must - * trigger a rebuild. Compared against the live model after every model change - * to decide whether the cached prompt is stale. - */ - #promptModelKey: string | undefined; - #builtInToolNames = new Set(); - #rpcHostToolNames = new Set(); - /** Session-owned `xd://` device registry (built-ins + dynamic mounts); `undefined` when the transport is off. */ - #xdevRegistry: XdevRegistry | undefined; - /** Names of discoverable tools currently mounted under `xd://` (dynamic mounts only, not built-in devices). */ - #mountedXdevToolNames = new Set(); - /** Coalesced xd:// mount delta not yet announced to the model; delivered as a - * hidden notice alongside the next prompt (see {@link #notifyXdevMountDelta}). */ - #pendingXdevMountDelta: { added: Set; removed: Set } | undefined; - // TTSR manager for time-traveling stream rules - #ttsrManager: TtsrManager | undefined = undefined; - #pendingTtsrInjections: Rule[] = []; - /** Per-tool TTSR rules whose `interruptMode` opted out of aborting the stream. - * These are folded into the matched tool call's `toolResult` content as an - * in-band system reminder, instead of spawning a separate follow-up turn. */ - #perToolTtsrInjections = new Map(); - #ttsrAbortPending = false; - #ttsrRetryToken = 0; - #ttsrResumePromise: Promise | undefined = undefined; - #ttsrResumeResolve: (() => void) | undefined = undefined; + readonly #ttsr: TtsrCoordinator; + readonly #stats: SessionStatsTracker; /** One-shot flag for expected internal plan-mode aborts. Approval actions may * abort the post-approval continuation before compaction, execution, or @@ -2207,20 +579,8 @@ export class AgentSession { #postPromptTasksResolve: (() => void) | undefined = undefined; #postPromptTasksAbortController = new AbortController(); - #streamingEditAbortTriggered = false; - #streamingEditCheckedLineCounts = new Map(); - - #streamingEditPrecheckedToolCallIds = new Set(); - - #streamingEditFileCache = new Map(); - - /** Active Gemini reasoning-header runaway detector for the current block. - * (Re)created on each `thinking_start` when the guard applies (see - * `#geminiHeaderGuardActive`); undefined for non-Gemini models or when the - * guard is off. Fed thinking deltas in the assistant-message interceptor. */ - #geminiHeaderDetector: GeminiHeaderRunDetector | undefined; - #toolCallLoopGuard: ToolCallLoopGuard | undefined; - #toolCallLoopGuardSettingsKey: string | undefined; + readonly #streamingEditGuard: StreamingEditGuard; + readonly #loopGuards: LoopGuards; #promptInFlightCount = 0; #abortInProgress = false; // Wire-level agent_end emission deferred until #promptInFlightCount drops to 0. @@ -2229,25 +589,10 @@ export class AgentSession { // `#emit(event)` that reaches external subscribers (rpc-mode stdout, ACP bridge, // Cursor exec, TUI listeners) is held back. Without this, a client that resumes // on `agent_end` can fire its next `prompt` before #promptWithMessage's finally - #emptyStopRetryCount = 0; - #unexpectedStopRetryCount = 0; - #acceptTerminalEmptyStopForPrompt = false; #promptGeneration = 0; #pendingAgentEndEmit: AgentSessionEvent | undefined; - #pendingContextSnapshot: - | { - promptTokens: number; - nonMessageTokens: number; - cutoffCount: number; - } - | undefined = undefined; #sessionStopContinuationCount = 0; #sessionStopHookActive = false; - // Bumped whenever the pending in-flight snapshot is set/cleared. The - // status-line context memo includes this so clearing the snapshot on - // turn-end/abort invalidates the cache even though the message list is - // unchanged — otherwise a mid-turn estimate would survive into idle. - #contextUsageRevision = 0; #obfuscator: SecretObfuscator | undefined; /** Session-start value of `inlineToolDescriptors`; drives handoff tool pruning. */ #pruneToolDescriptions = false; @@ -2266,18 +611,12 @@ export class AgentSession { #synchronouslyTerminatedYieldToolCallIds = new Set(); #providerSessionState = new Map(); #hindsightSessionState: HindsightSessionState | undefined = undefined; - #memoryAgentDir: string | undefined; - #memoryTaskDepth = 0; - #createMemoryTools: (() => Promise) | undefined; - #memoryBackendTransition: Promise = Promise.resolve(); - #localMemoryStartupAbort: AbortController | undefined; + readonly #memory: SessionMemory; readonly rawSseDebugBuffer: RawSseDebugBuffer; #resetPromptMaintenanceState(): void { - this.#emptyStopRetryCount = 0; - this.#unexpectedStopRetryCount = 0; + this.#recovery.resetForNewPrompt(); this.#yieldTerminationPending = false; - this.#acceptTerminalEmptyStopForPrompt = false; } #acquirePowerAssertion(): void { @@ -2351,7 +690,7 @@ export class AgentSession { // so they neither auto-resume the run the user stopped (a non-empty steer queue // otherwise bypasses the latch in #canAutoContinueForFollowUp) nor linger to // flush at the next prompt. Real user steers/follow-ups are left untouched. - if (this.#advisorAutoResumeSuppressed && !this.isStreaming) { + if (this.#advisors.autoResumeSuppressed && !this.isStreaming) { for (const card of this.#extractQueuedAdvisorCards()) { this.#preserveAdvisorCard(card); } @@ -2366,12 +705,9 @@ export class AgentSession { * the agent responds to the peer. Skip only when a queued steer/follow-up will itself drive a * resume turn whose aside poll already consumes these (no double-wake). */ #resumeStrandedIrcAsides(): void { - if (this.#isDisposed || this.isStreaming) return; - if (this.#pendingIrcInterrupts.length === 0 && this.#pendingIrcAsides.length === 0) return; + if (this.#isDisposed || this.isStreaming || !this.#irc.hasPending()) return; if (this.#canAutoContinueForFollowUp() && this.agent.hasQueuedMessages()) return; - const records = [...this.#pendingIrcInterrupts, ...this.#pendingIrcAsides]; - this.#pendingIrcInterrupts = []; - this.#pendingIrcAsides = []; + const records = this.#irc.drainPending(); if (this.#planModeState?.enabled) { // Plan mode: fold stranded IRC asides into context without waking an // autonomous turn. Convergence to ask/resolve stays user-driven. @@ -2474,178 +810,11 @@ export class AgentSession { this.#emit(pending); } - /** Advance the one-way prewalk switch at a completed assistant-turn boundary. */ - async #advancePrewalk(liveMessages: AgentMessage[], context: AgentTurnEndContext | undefined): Promise { - const prewalk = this.#prewalk; - if (!prewalk || context?.message.role !== "assistant") return; - - const todoSucceededThisTurn = context.toolResults.some(result => result.toolName === "todo" && !result.isError); - if (todoSucceededThisTurn) { - this.#prewalkTodoSeen = true; - } - - // The plan nudge asks for a prose plan before implementation begins, - // but the agent loop treats each text-only reply as terminal — observed - // silently killing production SWE-bench runs before any code was ever - // written. Tool progress re-arms one continuation, allowing split flows - // such as plan → todo → prose → read → prose → edit/write. Consuming the - // arm before steering also detects completion: two consecutive text-only - // replies have no intervening progress, so the second ends naturally - // instead of producing the #5551 loop. - const hasToolResults = context.toolResults.length > 0; - if (this.#prewalkPlanInjected && hasToolResults) { - this.#prewalkContinuePending = true; - } else if (this.#prewalkContinuePending) { - this.#prewalkContinuePending = false; - this.agent.steer({ - role: "custom", - customType: PREWALK_CONTINUE_MESSAGE_TYPE, - content: prewalkContinuePrompt, - attribution: "agent", - display: false, - timestamp: Date.now(), - }); - } - - // Todo gate: the plan nudge instructs "finish the plan, then init the - // todo list from it and start" — so the switch waits until a todo list - // exists AND the model has actually started implementing (first - // edit/write). The todo call itself never triggers: firing there handed - // the fast model the whole implementation cold. A failed todo call does - // not establish the list. The gate keys on the ACTIVE tool set, not the - // registry: a registered-but-deactivated todo (e.g. a restricted - // active-tool slate) is uncallable and would deadlock the switch. - const todoGateOpen = this.#prewalkTodoSeen || !this.getActiveToolNames().includes("todo"); - const action = todoGateOpen - ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) - : undefined; - if (!action) { - if (!this.#prewalkPlanInjected) { - this.#prewalkPlanInjected = true; - this.#prewalkContinuePending = true; - this.agent.steer({ - role: "custom", - customType: PREWALK_PLAN_MESSAGE_TYPE, - content: prewalkPlanPrompt, - display: false, - attribution: "agent", - timestamp: Date.now(), - }); - this.emitNotice("info", "Prewalk: injected deep-plan nudge.", "prewalk"); - } - return; - } - - await this.#waitForSessionMessagePersistence(context.message); - for (const toolResult of context.toolResults) { - await this.#waitForSessionMessagePersistence(toolResult); - } - - this.#scrubPrewalkPlanNudge(liveMessages); - const target = prewalk.target; - if (this.model && modelsAreEqual(this.model, target)) { - this.#prewalk = undefined; - return; - } - - await this.setModelTemporary(target, prewalk.thinkingLevel, { ephemeral: true }); - this.#prewalk = undefined; - this.emitNotice( - "info", - `Prewalk: switched to ${target.provider}/${target.id} after first ${action.toolName} call.`, - "prewalk", - ); - this.agent.steer({ - role: "custom", - customType: PREWALK_CHECKLIST_MESSAGE_TYPE, - content: prewalkChecklistPrompt, - attribution: "agent", - display: false, - timestamp: Date.now(), - }); - } - /** - * Arm prewalk outside the normal startup path (the `/prewalk` slash - * command): sets the target and immediately steers the plan nudge rather - * than waiting for the next turn boundary, since an explicit manual - * invocation means "start this now." A no-op with a notice if a prewalk - * is already armed and waiting. + * Arm prewalk outside the normal startup path so an explicit slash command starts immediately. */ armPrewalk(target: Model, thinkingLevel?: ConfiguredThinkingLevel): void { - if (this.#prewalk) { - this.emitNotice( - "info", - `Prewalk: already armed for ${this.#prewalk.target.provider}/${this.#prewalk.target.id}, waiting for the first edit/write.`, - "prewalk", - ); - return; - } - this.#prewalk = { target, thinkingLevel }; - this.#prewalkPlanInjected = true; - this.#prewalkContinuePending = true; - this.agent.steer({ - role: "custom", - customType: PREWALK_PLAN_MESSAGE_TYPE, - content: prewalkPlanPrompt, - display: false, - attribution: "agent", - timestamp: Date.now(), - }); - this.emitNotice( - "info", - `Prewalk: armed for ${target.provider}/${target.id} — will switch at the first edit/write once the todo list exists.`, - "prewalk", - ); - } - - /** - * Remove the plan nudge from the LLM context before the model switch: the - * fast model inherits the plan the nudge produced, not the nudge itself. - * Splices the loop's live context array in place (the run streams from - * it) and mirrors the removal into agent state. The persisted transcript - * keeps the message for audit; a session reload re-materializes it, - * which is acceptable for prewalk's single-run lifecycle. - */ - #scrubPrewalkPlanNudge(liveMessages: AgentMessage[]): void { - if (!this.#prewalkPlanInjected) return; - const isPlanNudge = (m: AgentMessage): boolean => - m.role === "custom" && m.customType === PREWALK_PLAN_MESSAGE_TYPE; - for (let i = liveMessages.length - 1; i >= 0; i--) { - if (isPlanNudge(liveMessages[i])) { - // Interior removal on the live array: drop the scrubbed message from - // the convert/estimate caches so the next convert can't reuse a prefix - // that still carries its fragment (the array shrinks in place). - invalidateMessageCache(liveMessages[i]); - liveMessages.splice(i, 1); - } - } - const stateMessages = this.agent.state.messages; - const filtered = stateMessages.filter(m => !isPlanNudge(m)); - if (filtered.length !== stateMessages.length) this.agent.replaceMessages(filtered); - } - - /** - * Lazily arm PlanYolo before the first prompt is built: restricts tools to - * the plan-mode read-only set (plus `write`, which carries the resolution - * devices including `xd://propose`), marks plan-mode state so - * `#buildPlanModeMessage` injects the standard plan-mode-active instructions - * on this and every following prompt, and registers the auto-approve plan-proposal handler. - * Idempotent — a no-op once armed or when PlanYolo is not configured. - */ - async #armPlanYoloIfNeeded(): Promise { - if (!this.#planYolo || this.#planYoloArmed) return; - this.#planYoloArmed = true; - const previousTools = this.getEnabledToolNames(); - const augmentations = this.hasBuiltInTool("write") ? ["write"] : []; - await this.setActiveToolsByName([...new Set([...previousTools, ...augmentations])]); - this.#planYoloPreviousTools = previousTools; - this.setPlanModeState({ - enabled: true, - planFilePath: this.getPlanReferencePath() || "local://PLAN.md", - workflow: "parallel", - }); - this.setPlanProposalHandler(title => this.#approvePlanYoloProposal(title)); + this.#prewalk.arm(target, thinkingLevel); } /** Validate the active plan artifact and shape an `xd://propose` result for review-mode hosts. */ @@ -2666,168 +835,197 @@ export class AgentSession { }; } - /** - * Plan-proposal handler while PlanYolo's plan phase is active. Auto-approves - * the instant the model writes the plan slug/title to `xd://propose` — no - * interactive review, the headless counterpart to plan mode's "Approve and - * execute" — then restores tools, exits plan-mode state, switches to the - * configured `target`, and hands off the approved plan for it to implement. - */ - #approvePlanYoloProposal(title: string): Promise> { - return this.#finalizePlanYoloProposal(title); - } - - async #finalizePlanYoloProposal(title: string): Promise> { - const planYolo = this.#planYolo; - const state = this.getPlanModeState(); - if (!planYolo || !state?.enabled) { - throw new ToolError("Plan mode is not active."); - } - const { planFilePath, title: resolvedTitle } = await resolveApprovedPlan({ - suppliedTitle: title, - statePlanFilePath: state.planFilePath, - readPlan: url => this.#readPlanFile(url), - listPlanFiles: () => this.#listPlanFiles(), - }); - this.setPlanModeState(undefined); - const previousTools = this.#planYoloPreviousTools; - try { - if (previousTools) { - await this.setActiveToolsByName(previousTools); - } - } catch (error) { - this.setPlanModeState(state); - throw error; - } - this.setPlanProposalHandler(null); - this.#planYolo = undefined; - this.#planYoloPreviousTools = undefined; - await this.setModelTemporary(planYolo.target, planYolo.thinkingLevel, { ephemeral: true }); - this.emitNotice( - "info", - `Plan-yolo: plan approved, switched to ${planYolo.target.provider}/${planYolo.target.id} to implement "${resolvedTitle}".`, - "plan-yolo", - ); - this.agent.steer({ - role: "custom", - customType: PLAN_YOLO_HANDOFF_MESSAGE_TYPE, - content: prompt.render(planYoloHandoffPrompt, { planFilePath, title: resolvedTitle }), - attribution: "agent", - display: false, - timestamp: Date.now(), - }); - return { - content: [{ type: "text" as const, text: `Plan approved. Implementing now with ${planYolo.target.id}.` }], - details: { planFilePath, title: resolvedTitle, planExists: true }, - }; - } - async #readPlanFile(planFilePath: string): Promise { - const resolvedPath = planFilePath.startsWith("local:") - ? resolveLocalUrlToPath(normalizeLocalScheme(planFilePath), this.#localProtocolOptions()) - : resolveToCwd(planFilePath, this.sessionManager.getCwd()); - try { - return await Bun.file(resolvedPath).text(); - } catch (error) { - if (isEnoent(error)) return null; - throw error; - } + return readPlanFile(planFilePath, { + localProtocolOptions: this.#localProtocolOptions(), + cwd: this.sessionManager.getCwd(), + }); } /** `local://` URLs of plan files in the session-local root, newest first — * a fallback for `resolveApprovedPlan` when the agent dropped `extra.title`. */ async #listPlanFiles(): Promise { - const localRoot = resolveLocalUrlToPath("local://", this.#localProtocolOptions()); - try { - const entries = await fs.promises.readdir(localRoot, { withFileTypes: true }); - const plans = await Promise.all( - entries - .filter(entry => entry.isFile() && /plan\.md$/i.test(entry.name)) - .map(async entry => { - const stat = await fs.promises.stat(path.join(localRoot, entry.name)).catch(() => null); - return { url: `local://${entry.name}`, mtime: stat?.mtimeMs ?? 0 }; - }), - ); - return plans.sort((a, b) => b.mtime - a.mtime).map(plan => plan.url); - } catch { - return []; - } + return listPlanFiles({ localProtocolOptions: this.#localProtocolOptions() }); } constructor(config: AgentSessionConfig) { this.agent = config.agent; this.sessionManager = config.sessionManager; - this.#bashSessionTarget = { - sessionId: this.sessionManager.getSessionId(), - refs: 0, - destination: { kind: "current", manager: this.sessionManager }, - }; this.settings = config.settings; - this.#autoApprove = config.autoApprove === true; + this.#modelRegistry = config.modelRegistry; + const bashHost: BashRunnerHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + extensionRunner: () => this.#extensionRunner, + isStreaming: () => this.isStreaming, + }; + this.#bash = new BashRunner(bashHost); // Power assertions are taken per turn (see #beginInFlight); nothing acquired here. - this.#evalKernelOwnerId = config.evalKernelOwnerId ?? `agent-session:${Snowflake.next()}`; - this.#parentEvalSessionId = config.parentEvalSessionId; + const evalHost: EvalRunnerHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + extensionRunner: () => this.#extensionRunner, + isStreaming: () => this.isStreaming, + appendSessionMessage: message => { + this.agent.appendMessage(message); + this.sessionManager.appendMessage(message); + }, + }; + this.#eval = new EvalRunner(evalHost, { + kernelOwnerId: config.evalKernelOwnerId ?? `agent-session:${Snowflake.next()}`, + parentSessionId: config.parentEvalSessionId, + }); + const ircHost: IrcBridgeHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + isDisposed: () => this.#isDisposed, + isStreaming: () => this.isStreaming, + planModeEnabled: () => this.#planModeState?.enabled === true, + emitSessionEvent: event => this.#emitSessionEvent(event), + wakeForIrc: records => this.#wakeForIrc(records), + runEphemeralTurn: args => this.runEphemeralTurn(args), + }; + this.#irc = new IrcBridge(ircHost); + const prewalkHost: PrewalkCoordinatorHost = { + agent: this.agent, + sessionManager: this.sessionManager, + model: () => this.model, + emitNotice: (level, message, source) => this.emitNotice(level, message, source), + setModelTemporary: (model, thinkingLevel, options) => this.setModelTemporary(model, thinkingLevel, options), + setActiveToolsByName: names => this.setActiveToolsByName(names), + getActiveToolNames: () => this.getActiveToolNames(), + getEnabledToolNames: () => this.getEnabledToolNames(), + hasBuiltInTool: name => this.hasBuiltInTool(name), + getPlanModeState: () => this.getPlanModeState(), + setPlanModeState: state => this.setPlanModeState(state), + getPlanReferencePath: () => this.getPlanReferencePath(), + setPlanProposalHandler: handler => this.setPlanProposalHandler(handler), + waitForSessionMessagePersistence: message => this.#waitForSessionMessagePersistence(message), + localProtocolOptions: () => this.#localProtocolOptions(), + }; + this.#prewalk = new PrewalkCoordinator(prewalkHost, { + prewalk: config.prewalk, + planYolo: config.planYolo, + }); + const todoHost: TodoTrackerHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + model: () => this.model, + agentKind: () => this.#agentKind, + emitSessionEvent: event => this.#emitSessionEvent(event), + scheduleAgentContinue: options => this.#scheduleAgentContinue(options), + promptGeneration: () => this.#promptGeneration, + hasPendingAsyncWake: () => this.#hasPendingAsyncWake(), + getActiveToolNames: () => this.getActiveToolNames(), + toolRegistry: () => this.#tools.registry, + planModeEnabled: () => this.#planModeState?.enabled === true, + consumeLastServedToolChoiceLabel: () => this.#toolChoiceQueue.consumeLastServedLabel(), + }; + this.#todo = new TodoTracker(todoHost); this.#ownedAsyncJobManager = config.ownedAsyncJobManager; this.#asyncJobManager = config.asyncJobManager ?? config.ownedAsyncJobManager; - this.#scopedModels = config.scopedModels ?? []; - if (config.thinkingLevel === AUTO_THINKING) { - // `auto` is session-level: keep the flag and show a provisional concrete - // level (the agent's initial effort was already set by the caller) until - // the first user turn is classified. - this.#autoThinking = true; - this.#thinkingLevel = resolveProvisionalAutoLevel(this.model); - } else { - this.#thinkingLevel = config.thinkingLevel; - } - if (config.initialRetryFallback) { - this.#activeRetryFallback = { - ...config.initialRetryFallback, - lastAppliedFallbackThinkingLevel: this.configuredThinkingLevel(), - pinned: false, - }; - } - if (config.prewalk) { - this.#prewalk = config.prewalk; - } - if (config.planYolo) { - this.#planYolo = config.planYolo; - } - this.#applyThinkingLevelToAgent(this.#thinkingLevel); + const modelControlsHost: ModelControlsHost = { + agent: this.agent, + settings: this.settings, + modelRegistry: this.#modelRegistry, + sessionManager: this.sessionManager, + providerSessionState: this.#providerSessionState, + model: () => this.model, + sessionId: () => this.sessionId, + promptGeneration: () => this.#promptGeneration, + resolveActiveEditMode: () => this.#tools.resolveActiveEditMode(), + syncAfterModelChange: previousEditMode => this.#tools.syncAfterModelChange(previousEditMode), + setModelWithProviderSessionReset: model => this.#setModelWithProviderSessionReset(model), + clearActiveRetryFallback: () => this.#recovery.clearActiveRetryFallback(), + clearInheritedProviderPromptCacheKey: () => this.#clearInheritedProviderPromptCacheKey(), + magicKeywordEnabled: keyword => this.#magicKeywordEnabled(keyword), + emit: event => this.#emit(event), + emitSessionEvent: event => this.#emitSessionEvent(event), + emitNotice: (level, message, source) => this.emitNotice(level, message, source), + }; + this.#models = new ModelControls(modelControlsHost, { + scopedModels: config.scopedModels, + thinkingLevel: config.thinkingLevel, + serviceTierByFamily: config.serviceTierByFamily, + }); this.#promptTemplates = config.promptTemplates ?? []; this.#slashCommands = config.slashCommands ?? []; this.#extensionRunner = config.extensionRunner; - this.#skills = config.skills ?? []; - this.#skillWarnings = config.skillWarnings ?? []; this.#customCommands = config.customCommands ?? []; - this.#skillsReloadable = config.skillsReloadable ?? true; - this.#skillsSettings = config.skillsSettings; - this.#modelRegistry = config.modelRegistry; - this.#memoryAgentDir = config.memoryAgentDir; - this.#memoryTaskDepth = config.memoryTaskDepth ?? 0; - this.#createMemoryTools = config.createMemoryTools; + const recoveryHost: TurnRecoveryHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + modelRegistry: this.#modelRegistry, + configWarnings: this.configWarnings, + model: () => this.model, + thinkingLevel: () => this.thinkingLevel, + configuredThinkingLevel: () => this.configuredThinkingLevel(), + setThinkingLevel: level => this.setThinkingLevel(level), + isDisposed: () => this.#isDisposed, + isStreaming: () => this.isStreaming, + isCompacting: () => this.isCompacting, + abortInProgress: () => this.#abortInProgress, + streamingEditAbortTriggered: () => this.#streamingEditGuard.abortTriggered, + promptGeneration: () => this.#promptGeneration, + sessionId: () => this.sessionId, + emitSessionEvent: event => this.#emitSessionEvent(event), + scheduleAgentContinue: options => this.#scheduleAgentContinue(options), + waitForSessionMessagePersistence: message => this.#waitForSessionMessagePersistence(message), + appendSessionMessage: message => this.#appendSessionMessage(message), + sessionMessageAlreadyPersisted: message => this.#sessionMessageAlreadyPersisted(message), + setModelWithProviderSessionReset: model => this.#setModelWithProviderSessionReset(model), + resetCurrentResponsesProviderSession: reason => this.#resetCurrentResponsesProviderSession(reason), + maybeAutoRedeemCodexReset: () => this.#maybeAutoRedeemCodexReset(), + runAutoCompaction: (reason, willRetry, deferred, allowDefer, options) => + this.#maintenance.runAutoCompaction(reason, willRetry, deferred, allowDefer, options), + withBashBranchTransition: operation => this.#bash.withBranchTransition(operation), + }; + this.#recovery = new TurnRecovery(recoveryHost, { initialRetryFallback: config.initialRetryFallback }); + const statsHost: SessionStatsTrackerHost = { + session: this, + agent: this.agent, + sessionManager: this.sessionManager, + modelRegistry: this.#modelRegistry, + model: () => this.model, + sessionId: () => this.sessionId, + }; + this.#stats = new SessionStatsTracker(statsHost); + const memoryHost: SessionMemoryHost = { + agent: this.agent, + settings: this.settings, + modelRegistry: this.#modelRegistry, + isDisposed: () => this.#isDisposed, + memoryBackendSession: () => this, + getHindsightSessionState: () => this.getHindsightSessionState(), + setHindsightSessionState: state => this.setHindsightSessionState(state), + getMnemopiSessionState: () => this.getMnemopiSessionState(), + takeMnemopiSessionState: () => setMnemopiSessionState(this, undefined), + setBaseSystemPrompt: prompt => { + this.#tools.setBaseSystemPrompt(prompt); + this.agent.setSystemPrompt(prompt); + }, + refreshBaseSystemPrompt: () => this.#tools.refreshBaseSystemPrompt(), + replaceMemoryTools: tools => this.#tools.replaceMemoryTools(tools), + }; + this.#memory = new SessionMemory(memoryHost, { + memoryAgentDir: config.memoryAgentDir, + memoryTaskDepth: config.memoryTaskDepth, + createMemoryTools: config.createMemoryTools, + }); // Resolve the wire service-tier per request so the Fireworks Priority // toggle scopes priority to Fireworks alone, without mutating the shared // session `serviceTier` that drives `/fast` and OpenAI/Anthropic priority. - this.agent.serviceTierResolver = model => this.#effectiveServiceTier(model); - this.#serviceTierByFamily = config.serviceTierByFamily ?? {}; - this.#advisorTools = config.advisorTools; - this.#advisorWatchdogPrompt = config.advisorWatchdogPrompt; - this.#advisorSharedInstructions = config.advisorSharedInstructions; - this.#advisorContextPrompt = config.advisorContextPrompt; - this.#advisorConfigs = config.advisorConfigs; + this.agent.serviceTierResolver = model => this.#models.effectiveServiceTier(model); this.#titleSystemPrompt = config.titleSystemPrompt; this.#pruneToolDescriptions = config.pruneToolDescriptions === true; - this.#validateRetryFallbackChains(); - this.#toolRegistry = config.toolRegistry ?? new Map(); - this.#createVibeTools = config.createVibeTools; - this.#builtInToolNames = new Set(config.builtInToolNames ?? []); - this.#presentationPinnedToolNames = config.presentationPinnedToolNames; - this.#ensureWriteRegistered = config.ensureWriteRegistered; this.#transformContext = config.transformContext ?? (messages => messages); - this.#transformProviderContext = config.transformProviderContext; this.#sideStreamFn = config.sideStreamFn ?? streamSimple; - this.#advisorStreamFn = config.advisorStreamFn; this.#preferWebsockets = config.preferWebsockets; this.#onPayload = config.onPayload; this.rawSseDebugBuffer = config.rawSseDebugBuffer ?? new RawSseDebugBuffer(); @@ -2839,12 +1037,12 @@ export class AgentSession { this.#onResponse = configuredOnResponse ? async (response, model) => { this.rawSseDebugBuffer.recordResponse(response, model); - this.#ingestProviderUsageHeaders(response, model); + this.#stats.ingestProviderUsageHeaders(response, model); await configuredOnResponse(response, model); } : (response, model) => { this.rawSseDebugBuffer.recordResponse(response, model); - this.#ingestProviderUsageHeaders(response, model); + this.#stats.ingestProviderUsageHeaders(response, model); }; const configuredOnSseEvent = config.onSseEvent; this.#onSseEvent = configuredOnSseEvent @@ -2864,38 +1062,10 @@ export class AgentSession { this.#pendingRewindReport = undefined; await this.#applyRewind(rewindReport, messages); } - if (context?.message.role === "assistant") { - const detection = this.#activeToolCallLoopGuard()?.recordTurn({ - message: context.message, - toolResults: context.toolResults, - }); - if (detection) this.#maybeInjectToolCallLoopRedirect(messages, detection); - } - await this.#advancePrewalk(messages, context); - this.#advisorPrimaryTurnsCompleted++; - if (this.#advisors.length > 0) { - for (const a of this.#advisors) { - if (a.runtime.disposed) continue; - try { - a.runtime.onTurnEnd(messages, { willContinue: context?.willContinue }); - } catch (advisorErr) { - // CRITICAL boundary: NOTHING an advisor does may abort the - // primary agent's turn-end. A throwing advisor loses its - // delta; the primary continues untouched. - logger.warn("advisor onTurnEnd threw; delta dropped", { - advisor: a.name, - err: String(advisorErr), - }); - } - } - const syncBacklog = this.settings.get("advisor.syncBacklog"); - if (syncBacklog !== "off") { - const threshold = parseInt(syncBacklog, 10); - // Parallel so the 30s catch-up budget is shared across advisors, not summed. - await Promise.all(this.#advisors.map(a => a.runtime.waitForCatchup(30000, threshold, signal))); - } - } - await this.#maintainContextMidRun(messages, signal, context); + this.#loopGuards.recordTurn(messages, context); + await this.#prewalk.advanceAtTurnEnd(messages, context); + await this.#advisors.onPrimaryTurnEnd(messages, context?.willContinue, signal); + await this.#maintenance.maintainContextMidRun(messages, signal, context); }); this.yieldQueue = new YieldQueue({ isStreaming: () => this.isStreaming, @@ -2917,31 +1087,100 @@ export class AgentSession { // each step boundary as non-interrupting asides. Peer IRCs share the aside // injection boundary, but also expose a non-consuming interrupt peek so // `hub` waits can return early before the boundary drains them. - this.agent.hasIrcInterrupts = () => this.#pendingIrcInterrupts.length > 0; + this.agent.hasIrcInterrupts = () => this.#irc.hasInterrupts(); this.agent.setAsideMessageProvider(() => { - const pendingIrc = [...this.#pendingIrcInterrupts, ...this.#pendingIrcAsides]; - this.#pendingIrcInterrupts = []; - this.#pendingIrcAsides = []; - const thunks: AsideMessage[] = pendingIrc.map(record => () => record); + const thunks: AsideMessage[] = this.#irc.drainPending().map(record => () => record); thunks.push(...this.yieldQueue.drainLazy()); // Mid-run todo reconciliation — evaluated at injection time so a turn // that flips a todo just before this poll suppresses the nudge. - thunks.push(() => this.#takeMidRunTodoNudge()); + thunks.push(() => this.#todo.takeMidRunNudge()); return thunks; }); this.#convertToLlm = config.convertToLlm ?? convertToLlm; - this.#rebuildSystemPrompt = config.rebuildSystemPrompt; - this.#getLocalCalendarDate = config.getLocalCalendarDate ?? formatLocalCalendarDate; - this.#getMcpServerInstructions = config.getMcpServerInstructions; this.getXdevToolEntries = config.getXdevToolEntries ?? (() => []); - this.#xdevRegistry = config.xdevRegistry; - this.#mountedXdevToolNames = new Set(config.initialMountedXdevToolNames ?? []); - this.#setActiveToolNames = config.setActiveToolNames; + const sessionToolsHost: SessionToolsHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + modelRegistry: this.#modelRegistry, + extensionRunner: () => this.#extensionRunner, + clientBridge: () => this.#clientBridge, + agentKind: () => this.#agentKind, + isDisposed: () => this.#isDisposed, + isStreaming: () => this.isStreaming, + queuedMessageCount: () => this.queuedMessageCount, + planModeEnabled: () => this.#planModeState?.enabled === true, + model: () => this.model, + memoryBackendSession: () => this, + clearInheritedProviderPromptCacheKey: () => this.#clearInheritedProviderPromptCacheKey(), + clearMemoryPromotionSnapshot: () => this.#memory.clearPromotionSnapshot(), + captureMemoryPromotionSnapshot: prompt => this.#memory.capturePromotionSnapshot(prompt), + emitNotice: (level, message, source) => this.emitNotice(level, message, source), + notifyCommandMetadataChanged: () => this.#notifyCommandMetadataChanged(), + localProtocolOptions: () => this.#localProtocolOptions(), + }; + this.#tools = new SessionTools(sessionToolsHost, { + autoApprove: config.autoApprove, + toolRegistry: config.toolRegistry, + createVibeTools: config.createVibeTools, + builtInToolNames: config.builtInToolNames, + presentationPinnedToolNames: config.presentationPinnedToolNames, + ensureWriteRegistered: config.ensureWriteRegistered, + rebuildSystemPrompt: config.rebuildSystemPrompt, + getLocalCalendarDate: config.getLocalCalendarDate, + getMcpServerInstructions: config.getMcpServerInstructions, + xdevRegistry: config.xdevRegistry, + initialMountedXdevToolNames: config.initialMountedXdevToolNames, + setActiveToolNames: config.setActiveToolNames, + baseSystemPrompt: this.agent.state.systemPrompt, + skills: config.skills, + skillWarnings: config.skillWarnings, + skillsSettings: config.skillsSettings, + skillsReloadable: config.skillsReloadable, + }); this.#disconnectOwnedMcpManager = config.disconnectOwnedMcpManager; - this.#baseSystemPrompt = this.agent.state.systemPrompt; - this.#promptModelKey = this.#currentPromptModelKey(); - this.#ttsrManager = config.ttsrManager; + const ttsrHost: TtsrCoordinatorHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + emitSessionEvent: event => this.#emitSessionEvent(event), + schedulePostPromptTask: (task, options) => this.#schedulePostPromptTask(task, options), + scheduleAgentContinue: options => this.#scheduleAgentContinue(options), + promptGeneration: () => this.#promptGeneration, + }; + this.#ttsr = new TtsrCoordinator(ttsrHost, config.ttsrManager); this.#obfuscator = config.obfuscator; + const providerBoundaryHost: SessionProviderBoundaryHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + modelRegistry: this.#modelRegistry, + model: () => this.model, + sessionId: () => this.sessionId, + localProtocolOptions: () => this.#localProtocolOptions(), + transformContext: (messages, signal) => this.#transformContext(messages, signal), + convertToLlm: messages => this.#convertToLlm(messages), + onPayload: this.#onPayload, + onResponse: this.#onResponse, + onSseEvent: this.#onSseEvent, + obfuscator: this.#obfuscator, + }; + this.#providerBoundary = new SessionProviderBoundary(providerBoundaryHost); + const streamGuardsHost: StreamGuardsHost = { + agent: this.agent, + settings: this.settings, + sessionManager: this.sessionManager, + obfuscator: this.#obfuscator, + model: () => this.model, + isDisposed: () => this.#isDisposed, + promptGeneration: () => this.#promptGeneration, + localProtocolOptions: () => this.#localProtocolOptions(), + emitNotice: (level, message, source) => this.emitNotice(level, message, source), + schedulePostPromptTask: task => this.#schedulePostPromptTask(task), + discardAssistantTurn: message => this.#recovery.discardAssistantTurn(message), + }; + this.#streamingEditGuard = new StreamingEditGuard(streamGuardsHost); + this.#loopGuards = new LoopGuards(streamGuardsHost); this.#agentId = config.agentId; this.#agentKind = config.agentKind ?? "main"; this.#providerSessionId = config.providerSessionId; @@ -2968,15 +1207,15 @@ export class AgentSession { message, assistantMessageEvent, }; - this.#preCacheStreamingEditFile(event); - this.#maybeAbortStreamingEdit(event); - this.#maybeInterruptGeminiHeaderRunaway(message, assistantMessageEvent); + this.#streamingEditGuard.preCache(event); + this.#streamingEditGuard.maybeAbort(event); + this.#loopGuards.onAssistantEvent(message, assistantMessageEvent); }); // Tool-result hook owns synchronous post-tool actions that must affect the current loop. this.agent.afterToolCall = ctx => this.#afterToolCall(ctx); this.agent.providerSessionState = this.#providerSessionState; this.#syncAgentSessionId(); - this.#syncTodoPhasesFromBranch(); + this.#todo.syncFromBranch(); this.#goalRuntime = new GoalRuntime({ getState: () => this.#goalModeState, setState: state => { @@ -3019,8 +1258,165 @@ export class AgentSession { this.#recordSessionExit(reason); }); - this.#advisorEnabled = this.settings.get("advisor.enabled") as boolean; - if (this.#advisorEnabled) this.#buildAdvisorRuntime(); + const advisorsHost: SessionAdvisorsHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + modelRegistry: this.#modelRegistry, + yieldQueue: this.yieldQueue, + obfuscator: this.#obfuscator, + providerSessionState: this.#providerSessionState, + preferWebsockets: this.#preferWebsockets, + onPayload: this.#onPayload, + onResponse: this.#onResponse, + onSseEvent: this.#onSseEvent, + agentKind: () => this.#agentKind, + isDisposed: () => this.#isDisposed, + abortInProgress: () => this.#abortInProgress, + allowAgentInitiatedTurns: () => this.#allowAcpAgentInitiatedTurns, + planModeState: () => this.#planModeState, + clientBridge: () => this.#clientBridge, + emitSessionEvent: event => this.#emitSessionEvent(event), + emitNotice: (level, message, source) => this.emitNotice(level, message, source), + sendCustomMessage: (message, options) => this.sendCustomMessage(message, options), + extractQueuedAdvisorCards: () => this.#extractQueuedAdvisorCards(), + dropPendingAdvisorCards: () => { + this.#pendingNextTurnMessages = this.#pendingNextTurnMessages.filter(message => !isAdvisorCard(message)); + }, + preserveAdvisorCard: card => this.#preserveAdvisorCard(card), + hasPendingNextTurnMessages: () => this.#pendingNextTurnMessages.length > 0, + convertToLlmForSideRequest: messages => this.#convertToLlmForSideRequest(messages), + effectiveServiceTier: model => this.#models.effectiveServiceTier(model), + resolveContextPromotionTarget: (model, contextWindow) => + this.#maintenance.resolveContextPromotionTarget(model, contextWindow), + resolveCompactionModelCandidates: (model, availableModels) => + this.#maintenance.resolveCompactionModelCandidates(model, availableModels), + resolveRetryFallbackRole: (selector, model) => this.#recovery.resolveRetryFallbackRole(selector, model), + findRetryFallbackCandidates: (role, selector, model) => + this.#recovery.findRetryFallbackCandidates(role, selector, model), + isRetryFallbackSelectorSuppressed: selector => this.#recovery.isRetryFallbackSelectorSuppressed(selector), + noteRetryFallbackCooldown: (selector, retryAfterMs, errorMessage) => + this.#recovery.noteRetryFallbackCooldown(selector, retryAfterMs, errorMessage), + createCodexCompactionContext: createMaintenanceCodexCompactionContext, + sessionId: () => this.sessionId, + }; + this.#advisors = new SessionAdvisors(advisorsHost, { + enabled: this.settings.get("advisor.enabled"), + tools: config.advisorTools, + watchdogPrompt: config.advisorWatchdogPrompt, + sharedInstructions: config.advisorSharedInstructions, + contextPrompt: config.advisorContextPrompt, + configs: config.advisorConfigs, + streamFn: config.advisorStreamFn, + transformProviderContext: config.transformProviderContext, + }); + + const maintenanceHost: SessionMaintenanceHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + modelRegistry: this.#modelRegistry, + extensionRunner: this.#extensionRunner, + sideStreamFn: this.#sideStreamFn, + providerSessionState: this.#providerSessionState, + model: () => this.model, + thinkingLevel: () => this.thinkingLevel, + isDisposed: () => this.#isDisposed, + isStreaming: () => this.isStreaming, + isGeneratingHandoff: () => this.isGeneratingHandoff, + promptGeneration: () => this.#promptGeneration, + sessionId: () => this.sessionId, + messages: () => this.messages, + baseSystemPrompt: () => this.#tools.baseSystemPrompt, + goalModeState: () => this.#goalModeState, + planReferencePath: () => this.#planReferencePath, + nonMessageTokenSource: () => this, + memoryBackendSession: () => this, + emitSessionEvent: event => this.#emitSessionEvent(event), + emitNotice: (level, message, source) => this.emitNotice(level, message, source), + schedulePostPromptTask: (task, options) => this.#schedulePostPromptTask(task, options), + scheduleAgentContinue: options => this.#scheduleAgentContinue(options), + scheduleCompactionContinuation: options => this.#scheduleCompactionContinuation(options), + persistTurnMessagesForMidRunCompaction: context => this.#persistTurnMessagesForMidRunCompaction(context), + findLastAssistantMessage: () => this.#findLastAssistantMessage(), + disconnectFromAgent: () => this.#disconnectFromAgent(), + reconnectToAgent: () => this.#reconnectToAgent(), + drainStrandedQueuedMessages: () => this.#drainStrandedQueuedMessages(), + buildDisplaySessionContext: () => this.buildDisplaySessionContext(), + convertToLlmForSideRequest: messages => this.#convertToLlmForSideRequest(messages), + obfuscateTextForProvider: text => this.#obfuscateTextForProvider(text), + obfuscatePreparationForProvider: preparation => this.#obfuscatePreparationForProvider(preparation), + closeCodexProviderSessionsForHistoryRewrite: () => this.#closeCodexProviderSessionsForHistoryRewrite(), + resetCodexProviderAfterCompaction: compaction => this.#resetCodexProviderAfterCompaction(compaction), + resetPlanReference: () => { + this.#planReferenceSent = false; + }, + syncTodoPhasesFromBranch: () => this.#todo.syncFromBranch(), + resetAdvisorRuntimes: () => this.#advisors.resetAllRuntimes(), + rebaseAfterCompaction: () => this.#stats.rebaseAfterCompaction(), + getContextBreakdown: options => this.getContextBreakdown(options), + getContextUsage: options => this.getContextUsage(options), + shake: (mode, options) => this.shake(mode, options), + dropImages: () => this.dropImages(), + runHandoff: (customInstructions, options) => this.handoff(customInstructions, options), + removeAssistantMessageFromActiveContext: message => + this.#recovery.removeAssistantMessageFromActiveContext(message), + dropPersistedAssistantTurn: message => this.#recovery.dropPersistedAssistantTurn(message), + runRecoveryCompactionWithRollback: (reason, message, allowDefer, options) => + this.#recovery.runRecoveryCompactionWithRollback(reason, message, allowDefer, options), + parseRetryAfterMsFromError: errorMessage => this.#recovery.parseRetryAfterMsFromError(errorMessage), + setModelTemporary: (model, thinkingLevel, options) => this.setModelTemporary(model, thinkingLevel, options), + abort: options => this.abort(options), + abortHandoff: () => this.abortHandoff(), + }; + this.#maintenance = new SessionMaintenance(maintenanceHost); + + const handoffHost: SessionHandoffHost = { + agent: this.agent, + sessionManager: this.sessionManager, + settings: this.settings, + modelRegistry: this.#modelRegistry, + extensionRunner: this.#extensionRunner, + sideStreamFn: this.#sideStreamFn, + obfuscator: this.#obfuscator, + model: () => this.model, + thinkingLevel: () => this.thinkingLevel, + sessionId: () => this.sessionId, + sessionFile: () => this.sessionFile, + baseSystemPrompt: () => this.#tools.baseSystemPrompt, + assertVibeSessionTransitionAllowed: action => this.#assertVibeSessionTransitionAllowed(action), + setSkipPostTurnMaintenance: timestamp => { + this.#maintenance.skipPostTurnMaintenanceAssistantTimestamp = timestamp; + }, + obfuscateTextForProvider: text => this.#obfuscateTextForProvider(text), + deobfuscateFromProvider: text => this.#deobfuscateFromProvider(text), + convertMessagesToLlm: (messages, signal) => this.convertMessagesToLlm(messages, signal), + prepareSimpleStreamOptions: (options, provider) => this.prepareSimpleStreamOptions(options, provider), + effectiveServiceTier: model => this.#models.effectiveServiceTier(model), + flushPendingBash: () => this.#bash.flushPending(), + beginBashSessionTransition: () => this.#bash.beginSessionTransition(), + markBashSessionTransition: transition => this.#bash.markSessionTransition(transition), + finishBashSessionTransition: (transition, success) => this.#bash.finishSessionTransition(transition, success), + cancelOwnAsyncJobs: () => this.#cancelOwnAsyncJobs(), + clearCheckpointRuntimeState: () => this.#clearCheckpointRuntimeState(), + clearFreshProviderSessionId: () => { + this.#freshProviderSessionId = undefined; + }, + syncAgentSessionId: () => this.#syncAgentSessionId(), + rekeyMemoryForCurrentSessionId: () => { + this.#memory.rekeyForCurrentSessionId(); + }, + resetMemoryContextForNewTranscript: () => this.#memory.resetContextForNewTranscript(), + clearPendingNextTurnMessages: () => { + this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; + }, + resetTodoCycle: () => this.#todo.resetCycle(), + buildDisplaySessionContext: () => this.buildDisplaySessionContext(), + resetAdvisorRuntimes: () => this.#advisors.resetAllRuntimes(), + syncTodoPhasesFromBranch: () => this.#todo.syncFromBranch(), + }; + this.#handoff = new SessionHandoff(handoffHost); this.#rehydrateCheckpointRewindState(); @@ -3029,928 +1425,8 @@ export class AgentSession { this.#unsubscribeAgent = this.agent.subscribe(this.#handleAgentEvent); // Re-evaluate append-only context mode when the setting changes at runtime. this.#unsubscribeAppendOnly = onAppendOnlyModeChanged(_value => this.#syncAppendOnlyContext(this.model)); - this.#unsubscribeModelRoles = onModelRolesChanged(() => { - if (!this.#advisorEnabled || this.#isDisposed) return; - if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime(); - this.#buildAdvisorRuntime(true); - }); + this.#unsubscribeModelRoles = onModelRolesChanged(() => this.#advisors.onModelRolesChanged()); } - // ------------------------------------------------------------------------- - // Advisor runtime lifecycle - // ------------------------------------------------------------------------- - #advisorImmuneTurnLimit(): number { - const immuneTurns = this.settings.get("advisor.immuneTurns") as number; - if (!Number.isFinite(immuneTurns) || immuneTurns <= 0) return 0; - return Math.trunc(immuneTurns); - } - - #isAdvisorInterruptImmuneTurnActive(): boolean { - return isAdvisorInterruptImmuneTurnActive({ - completedTurns: this.#advisorPrimaryTurnsCompleted, - immuneTurnStart: this.#advisorInterruptImmuneTurnStart, - immuneTurns: this.#advisorImmuneTurnLimit(), - }); - } - - // The next primary turn number starts the immune-turn window. While the - // interrupting steer is still in flight, completedTurns is lower than this - // start, so duplicate concern/blocker advice is also downgraded. - #recordAdvisorInterruptDelivered(): void { - this.#advisorInterruptImmuneTurnStart = this.#advisorPrimaryTurnsCompleted + 1; - } - - /** - * Re-prime the advisor across a conversation boundary: `/new`, `/branch`, - * `/btw`, `/tree`, and session switch/resume. Beyond {@link AdvisorRuntime.reset} - * (which only re-primes the advisor's transcript view and is also fired by - * within-conversation rewrites like compaction/shake/rewind), this clears the - * session-level interrupt latches so the prior conversation's cooldown cannot - * leak into the new one: the post-interrupt immune-turn window - * (`#advisorPrimaryTurnsCompleted`, `#advisorInterruptImmuneTurnStart`) and the - * user-interrupt auto-resume suppression flag. It also drops advisor deliveries - * still queued against the prior conversation — pending asides in the yield - * queue (advisor entries use `skipIdleFlush`, so they linger until the next - * `drainLazy` rather than self-flushing), interrupting cards parked in the - * agent steer/follow-up queues, and preserved cards deferred to the next turn — - * so none of them inject into the new conversation. - */ - #resetAdvisorSessionState(): void { - // Mute the recorder across the re-prime: AdvisorRuntime.reset() aborts the advisor - // loop, and that abort can emit an `aborted` message_end we must not attribute to - // either session's transcript. Detach, reset, then re-attach the live agent's feed. - for (const a of this.#advisors) { - a.agentUnsubscribe?.(); - a.agentUnsubscribe = undefined; - a.runtime.reset(); - a.adviseTool.resetDeliveredNotes(); - a.emissionGuard.reset(); - this.#attachAdvisorRecorderFeed(a); - } - this.#advisorPrimaryTurnsCompleted = 0; - this.#advisorInterruptImmuneTurnStart = undefined; - this.#advisorAutoResumeSuppressed = false; - this.yieldQueue.clear("advisor"); - this.#extractQueuedAdvisorCards(); - if (this.#pendingNextTurnMessages.some(isAdvisorCard)) { - this.#pendingNextTurnMessages = this.#pendingNextTurnMessages.filter(m => !isAdvisorCard(m)); - } - } - - #resolveAdvisorRuntimeDescriptors(emitWarnings: boolean): AdvisorRuntimeDescriptor[] { - const legacy = !this.#advisorConfigs?.length; - const roster: AdvisorConfig[] = legacy ? [{ name: "default" }] : this.#advisorConfigs!; - const descriptors: AdvisorRuntimeDescriptor[] = []; - const usedSlugs = new Set(); - for (const config of roster) { - let slug = legacy ? "" : slugifyAdvisorName(config.name); - if (slug) { - let candidate = slug; - let n = 2; - while (usedSlugs.has(candidate)) candidate = `${slug}-${n++}`; - slug = candidate; - usedSlugs.add(slug); - } - // Per-advisor toggle: skip disabled advisors but keep them in the - // status map so they show `○` rather than disappearing. - if (config.enabled === false) { - this.#advisorStatuses.set(slug, { name: config.name, status: "paused" }); - continue; - } - - // Resolve the advisor's model: an explicit `model` override wins; else the - // `advisor` role chain. A model that fails to resolve skips just this advisor. - let model: Model | undefined; - let thinkingLevel: ThinkingLevel | undefined; - if (config.model) { - const resolved = resolveModelOverride([config.model], this.#modelRegistry, this.settings); - model = resolved.model; - thinkingLevel = concreteThinkingLevel(resolved.thinkingLevel); - if (!model) { - this.#advisorStatuses.set(slug, { name: config.name, status: "no_model" }); - if (emitWarnings) { - this.emitNotice("warning", `Advisor "${config.name}": no model matched "${config.model}"`, "advisor"); - } - continue; - } - } else { - const sel = resolveAdvisorRoleSelection(this.settings, this.#modelRegistry.getAvailable()); - if (!sel) { - this.#advisorStatuses.set(slug, { name: config.name, status: "no_model" }); - if (emitWarnings) { - logger.debug("advisor enabled but no model assigned to the 'advisor' role; advisor inactive", { - advisor: config.name, - }); - } - continue; - } - model = sel.model; - thinkingLevel = concreteThinkingLevel(sel.thinkingLevel); - } - // Clamp the effort against the resolved model. Historically we defaulted - // to `ThinkingLevel.Medium` unconditionally, which threw at first stream - // on reasoning models that expose no controllable effort surface - // (e.g. `devin-agent`: Cascade routes by sibling model id, not a wire - // param; `getSupportedEfforts` returns `[]`). `resolveThinkingLevelForModel` - // preserves an explicit `off`, clamps a concrete effort into the model's - // supported range, and returns `undefined` for reasoning models without - // controllable efforts — for that case we forward `Inherit` so no effort - // is sent and reasoning stays enabled (matching the `auto`-path fix for - // Devin models via `clampAutoThinkingEffort`). See #4579. - const requestedLevel = thinkingLevel ?? ThinkingLevel.Medium; - const resolvedLevel = resolveThinkingLevelForModel(model, requestedLevel); - const advisorThinkingLevel: ThinkingLevel = resolvedLevel ?? ThinkingLevel.Inherit; - // Record the status entry now (in roster order) so the Map's insertion - // order matches the configured roster even when earlier advisors were - // skipped as paused/no_model. The build loop overwrites this to "running" - // without changing insertion order. - this.#advisorStatuses.set(slug, { name: config.name, status: "running" }); - descriptors.push({ - config, - name: config.name, - slug, - model, - thinkingLevel: advisorThinkingLevel, - signature: this.#advisorRuntimeSignature(config, slug, model, advisorThinkingLevel), - }); - } - return descriptors; - } - - #advisorRuntimeSignature(config: AdvisorConfig, slug: string, model: Model, thinkingLevel: ThinkingLevel): string { - const tools = config.tools?.length ? config.tools.join("\u001e") : ""; - const instructions = config.instructions?.trim() ?? ""; - return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join( - "\u001f", - ); - } - - #advisorRuntimeMatchesCurrentConfig(): boolean { - const descriptors = this.#resolveAdvisorRuntimeDescriptors(false); - if (descriptors.length !== this.#advisors.length) return false; - for (let i = 0; i < descriptors.length; i++) { - if (descriptors[i].signature !== this.#advisors[i].signature) return false; - } - return true; - } - - #buildAdvisorRuntime(seedToCurrent = false): boolean { - if (this.#isDisposed) return false; - if (this.#advisors.length > 0) return true; - if (!this.#advisorEnabled) return false; - if (this.#agentKind !== "main" && !this.settings.get("advisor.subagents")) return false; - - // Rebuild the status map from scratch so removed/renamed advisors don't - // leave stale entries. #resolveAdvisorRuntimeDescriptors populates every - // entry (`paused`/`no_model`/`running`) in roster order; the build loop - // below confirms `running` for successfully built advisors. - this.#advisorStatuses.clear(); - const descriptors = this.#resolveAdvisorRuntimeDescriptors(true); - - // Advisor service tier (`tier.advisor`): "none" (default) runs the advisor - // on standard processing; "inherit" tracks the session's live per-family - // tiers per request (like the main agent, including /fast toggles); a - // concrete value is broadcast across families and applied to the advisor - // model's family. One value for all advisors. - const advisorTierSetting = this.settings.get("tier.advisor"); - const advisorTierMap = - advisorTierSetting === "inherit" - ? undefined - : serviceTierForAllFamilies(serviceTierSettingToTier(advisorTierSetting)); - const advisorServiceTierResolver = (model: Model): ServiceTier | undefined => - advisorTierSetting === "inherit" - ? this.#effectiveServiceTier(model) - : resolveModelServiceTier(advisorTierMap, model); - - for (const descriptor of descriptors) { - const { - config, - slug, - model: advisorModel, - name: advisorName, - thinkingLevel: advisorThinkingLevel, - signature, - } = descriptor; - - const emissionGuard = new AdvisorEmissionGuard(); - const adviseTool = new AdviseTool((note, severity) => this.#routeAdvice(advisorRef, note, severity)); - - // `#advisorWatchdogPrompt` already carries WATCHDOG.md + YAML shared - // instructions; `config.instructions` adds this advisor's specialization. - const systemPrompt = [advisorSystemPrompt]; - if (this.#advisorContextPrompt) systemPrompt.push(this.#advisorContextPrompt); - if (this.#advisorWatchdogPrompt) systemPrompt.push(this.#advisorWatchdogPrompt); - if (this.#advisorSharedInstructions) systemPrompt.push(this.#advisorSharedInstructions); - if (config.instructions?.trim()) systemPrompt.push(config.instructions.trim()); - - const names = config.tools === undefined ? ADVISOR_DEFAULT_TOOL_NAMES : new Set(config.tools); - const tools = (this.#advisorTools ?? []).filter(t => names.has(t.name)); - const advisorLoopTools: AgentTool[] = [adviseTool, ...tools]; - const advisorToolMap = new Map(); - const availableAdvisorToolNames = new Set(); - for (const tool of advisorLoopTools) { - availableAdvisorToolNames.add(tool.name); - advisorToolMap.set(tool.name, tool); - if (tool.customWireName !== undefined) { - availableAdvisorToolNames.add(tool.customWireName); - advisorToolMap.set(tool.customWireName, tool); - } - } - let quarantinedAdvisorOutput: string | undefined; - let currentAdvisorInput = ""; - - const primaryProviderSessionId = this.sessionId; - const advisorSessionLabel = slug - ? `${primaryProviderSessionId}-advisor-${slug}` - : `${primaryProviderSessionId}-advisor`; - const advisorProviderSessionId = getOrCreateAdvisorProviderSessionId( - this.#advisorProviderSessionIds, - primaryProviderSessionId, - slug, - ); - const appendOnlyContext = new AppendOnlyContextManager(); - - // Thread the primary's telemetry into the advisor loop so the advisor - // model's GenAI spans + usage/cost hooks fire stamped with the local advisor - // identity. `conversationId` is cleared so provider telemetry falls back to - // the UUIDv7 provider session id, not the local `-advisor` label. - const advisorTelemetry = this.agent.telemetry - ? { - ...this.agent.telemetry, - agent: { - id: advisorSessionLabel, - name: slug ? `${MODEL_ROLES.advisor.name}: ${advisorName}` : MODEL_ROLES.advisor.name, - description: formatModelString(advisorModel), - }, - conversationId: undefined, - } - : undefined; - // Mirror the SDK's provider-shaping options (streamFn/onPayload/..., - // providerSessionState, promptCacheKey, transformProviderContext) so each - // advisor's requests cache, route, and obfuscate like the main turn. - // `promptCacheKey` preserves an explicitly pinned provider cache key - // unchanged so tan/shared-session advisor calls read the exact shard the - // parent turn populated. Otherwise the advisor uses its provider UUIDv7 so - // Codex request identity remains UUID-shaped while local labels keep the - // `-advisor` suffix. - const advisorPromptCacheKey = this.agent.promptCacheKey ?? advisorProviderSessionId; - // On the Cursor provider every tool runs server-side and is dispatched - // back through `cursorExecHandlers`; without this bridge the advisor's - // own tools (including the MCP `advise` tool) return `toolNotFound` and - // no advice is ever routed (issue #5680). Mirrors the primary agent's - // bridge (`sdk.ts`), scoped to this advisor's granted tool set. - // Cursor's native `delete` frame removes files directly, bypassing the - // tool map, so gate it on the advisor actually holding a file-mutating - // tool. A default read-only advisor (advise/read/grep/glob) never gets - // to delete workspace files it was never granted (issue #5680 review). - const advisorCanMutateFiles = advisorToolMap.has("write") || advisorToolMap.has("edit"); - if (advisorCanMutateFiles) availableAdvisorToolNames.add("delete"); - const advisorCursorExecHandlers = new CursorExecHandlers({ - cwd: this.sessionManager.getCwd(), - getCwd: () => this.sessionManager.getCwd(), - tools: advisorToolMap, - allowNativeDelete: advisorCanMutateFiles, - }); - const advisorAgent = new Agent({ - initialState: { - systemPrompt, - model: advisorModel, - thinkingLevel: toReasoningEffort(advisorThinkingLevel), - tools: advisorLoopTools, - }, - appendOnlyContext, - sessionId: advisorProviderSessionId, - promptCacheKey: advisorPromptCacheKey, - providerSessionState: this.#providerSessionState, - cursorExecHandlers: advisorCursorExecHandlers, - cwdResolver: () => this.sessionManager.getCwd(), - preferWebsockets: this.#preferWebsockets, - getApiKey: requestModel => this.#modelRegistry.resolver(requestModel, advisorProviderSessionId), - streamFn: this.#advisorStreamFn, - onPayload: this.#onPayload, - onResponse: this.#onResponse, - onSseEvent: this.#onSseEvent, - transformProviderContext: this.#transformProviderContext, - intentTracing: false, - transformAssistantMessage: message => { - quarantinedAdvisorOutput = quarantineAdvisorUnsafeOutput( - message, - availableAdvisorToolNames, - buildAdvisorQuarantineSourceText(currentAdvisorInput, advisorAgent.state.messages), - ); - }, - telemetry: advisorTelemetry, - serviceTier: undefined, - serviceTierResolver: advisorServiceTierResolver, - }); - advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel)); - - const advisorAgentFacade: AdvisorAgent = { - prompt: async input => { - let quarantined: string | undefined; - try { - quarantinedAdvisorOutput = undefined; - currentAdvisorInput = input; - await advisorAgent.prompt(input); - quarantined = quarantinedAdvisorOutput; - } finally { - quarantinedAdvisorOutput = undefined; - currentAdvisorInput = ""; - } - if (quarantined) throw new AdvisorOutputQuarantinedError(quarantined); - }, - abort: reason => advisorAgent.abort(reason), - reset: () => { - advisorAgent.reset(); - appendOnlyContext.log.clear(); - }, - rollbackTo: count => { - // Drop the failed user batch + synthetic assistant-error turn - // `Agent.#runLoop` appended for a turn ending in `stopReason: "error"`. - const messages = advisorAgent.state.messages; - if (count < messages.length) { - messages.length = count; - } - appendOnlyContext.resetSyncCursor(); - advisorAgent.state.error = undefined; - }, - state: advisorAgent.state, - }; - - // Persist this advisor's turns to `/__advisor[.].jsonl` - // (resolved lazily so it follows session switches) for stats attribution - // and Agent Hub observability, without registering it as a peer. - const recorder = new AdvisorTranscriptRecorder( - () => this.sessionManager.getSessionFile(), - () => this.sessionManager.getCwd(), - advisorTranscriptFilename(slug), - // On the advisor on→off→on toggle, wait for the prior recorders' closes - // so two SessionManagers never hold the same file at once. - this.#advisorRecorderClosed, - ); - const runtime = new AdvisorRuntime(advisorAgentFacade, { - snapshotMessages: () => this.agent.state.messages, - enqueueAdvice: (note, severity) => this.#routeAdvice(advisorRef, note, severity), - maintainContext: incomingTokens => this.#maintainAdvisorContext(advisorRef, incomingTokens), - obfuscator: this.#obfuscator, - beginAdvisorUpdate: () => advisorRef.emissionGuard.beginUpdate(), - onTurnError: (error, failedMessages) => this.#recoverAdvisorTurn(advisorRef, error, failedMessages), - onTurnSuccess: async () => { - const fallback = advisorRef.retryFallback; - if (!advisorRef.retryFallbackPendingSuccess || !fallback) return; - advisorRef.retryFallbackPendingSuccess = false; - await this.#emitSessionEvent({ - type: "retry_fallback_succeeded", - model: formatRetryFallbackSelector(advisorRef.agent.state.model, advisorRef.thinkingLevel), - role: fallback.role, - }); - }, - notifyFailure: error => { - this.#advisorStatuses.set(slug, { name: advisorName, status: "error" }); - const message = error instanceof Error ? error.message : String(error); - this.emitNotice( - "warning", - `Advisor${slug ? ` "${advisorName}"` : ""} unavailable for ${formatModelString(advisorAgent.state.model)}: ${message}`, - "advisor", - ); - }, - notifyQuotaExhausted: () => { - this.#advisorStatuses.set(slug, { name: advisorName, status: "quota_exhausted" }); - this.emitNotice("warning", `Advisor "${advisorName}" quota exhausted — pausing until reset.`, "advisor"); - }, - }); - - const advisorRef: ActiveAdvisor = { - name: advisorName, - slug, - agent: advisorAgent, - runtime, - adviseTool, - emissionGuard, - recorder, - recorderClosed: Promise.resolve(), - model: advisorModel, - thinkingLevel: advisorThinkingLevel, - providerSessionId: advisorProviderSessionId, - retryFallbackPendingSuccess: false, - signature, - }; - this.#attachAdvisorRecorderFeed(advisorRef); - if (seedToCurrent) runtime.seedTo(this.agent.state.messages.length); - this.#advisorStatuses.set(slug, { name: advisorName, status: "running" }); - this.#advisors.push(advisorRef); - } - - // One shared non-blocking aside channel for all advisors; the build callback - // aggregates every advisor's queued nits into one card (each entry already - // carries its own `advisor` name). - if (this.#advisors.length > 0 && !this.#advisorYieldQueueUnsubscribe) { - this.#advisorYieldQueueUnsubscribe = this.yieldQueue.register("advisor", { - build: entries => - entries.length === 0 - ? null - : ({ - role: "custom", - customType: "advisor", - display: true, - attribution: "agent", - timestamp: Date.now(), - content: formatAdvisorBatchContent(entries), - details: { notes: entries } satisfies AdvisorMessageDetails, - } satisfies CustomMessage), - skipIdleFlush: true, - }); - } - - return this.#advisors.length > 0; - } - - /** - * Route one accepted advice note from `advisor` to the primary. Concern and - * blocker interrupt the running agent through the steering channel; once the - * loop has yielded, `triggerTurn` resumes it. After a terminal text answer with - * no queued work, a concern is preserved as a visible advisor card, while a - * blocker wakes the primary to acknowledge work it handed off incorrectly. - * After a deliberate user interrupt auto-resume is suppressed while idle/unwinding - * (the note becomes a preserved card re-entering on resume); a live-streaming turn is - * steered in directly. A plain nit always rides the non-interrupting YieldQueue - * aside. Suppression by the per-advisor emission guard drops the note silently — - * the model still saw `Recorded.`, so it isn't tempted to rephrase the same note - * past the dedupe. - */ - #hasTerminalTextAnswerWithoutQueuedWork(): boolean { - if (this.agent.hasQueuedMessages() || this.#pendingNextTurnMessages.length > 0) return false; - const messages = this.agent.state.messages; - let tail = messages.length - 1; - while (tail >= 0 && isAdvisorCard(messages[tail])) tail--; - return isTerminalTextAssistantAnswer(messages[tail]); - } - - #routeAdvice(advisor: ActiveAdvisor, note: string, severity?: AdvisorSeverity): void { - if (!advisor.emissionGuard.accept(note)) { - logger.debug("advisor advice suppressed by emission guard", { severity, advisor: advisor.name }); - return; - } - // When newer primary turns already arrived while the advisor model was - // processing this batch, the advice was generated without seeing them. - // Append a lightweight staleness caveat so the primary can weigh recency. - const deliveredNote = annotateForStaleness(note, advisor.runtime.hasFreshBacklog); - // The implicit single ("default") advisor stamps no source name, so its - // agent-facing `` bytes stay identical to the pre-multi-advisor path. - const source = advisor.slug ? advisor.name : undefined; - const interrupting = isInterruptingSeverity(severity); - const channel = resolveAdvisorDeliveryChannel({ - severity, - autoResumeSuppressed: this.#advisorAutoResumeSuppressed, - preserveOnly: this.#preserveAdvisorAdvice, - // Key on the live agent-core loop, not session `isStreaming` (which also - // counts `#promptInFlightCount` during post-turn unwind). Only a running - // loop consumes a steer at its next boundary. - streaming: this.agent.state.isStreaming, - aborting: this.#abortInProgress, - terminalAnswerNoQueuedWork: this.#hasTerminalTextAnswerWithoutQueuedWork(), - interruptImmuneTurnActive: interrupting && this.#isAdvisorInterruptImmuneTurnActive(), - }); - if (channel === "aside") { - this.yieldQueue.enqueue("advisor", { note: deliveredNote, severity, advisor: source }); - return; - } - const notes: AdvisorNote[] = [{ note: deliveredNote, severity, advisor: source }]; - const content = formatAdvisorBatchContent(notes); - const details = { notes } satisfies AdvisorMessageDetails; - if (channel === "preserve") { - this.#preserveAdvisorCard({ - role: "custom", - customType: "advisor", - content, - display: true, - attribution: "agent", - details, - timestamp: Date.now(), - }); - return; - } - // A steered interrupting note only continues the run when the session can - // actually start (or is already running) a turn. Two idle cases cannot, so - // `sendCustomMessage({ triggerTurn: true })` would silently bury the card in - // `#pendingNextTurnMessages` until the next user prompt — strictly worse than - // the visible preserved card. Preserve instead: - // - Plan mode: only user-driven turns converge on ask/resolve. - // - ACP bridges with `deferAgentInitiatedTurns`: the client cannot show an - // agent-initiated turn as busy, so idle triggers are refused (#5628 review). - const cannotAutoTrigger = - !this.agent.state.isStreaming && - this.#clientBridge?.deferAgentInitiatedTurns === true && - !this.#allowAcpAgentInitiatedTurns; - if (this.#planModeState?.enabled || cannotAutoTrigger) { - this.#preserveAdvisorCard({ - role: "custom", - customType: "advisor", - content, - display: true, - attribution: "agent", - details, - timestamp: Date.now(), - }); - return; - } - // Arm the post-interrupt immune window only now that a turn is actually - // being steered/triggered. A merely preserved card never interrupts, so - // arming earlier would downgrade the next `advisor.immuneTurns` worth of - // real concerns/blockers to skip-idle-flush asides (#5628 review). - this.#recordAdvisorInterruptDelivered(); - void this.sendCustomMessage( - { customType: "advisor", content, display: true, attribution: "agent", details }, - { deliverAs: "steer", triggerTurn: true }, - ).catch(err => logger.debug("advisor delivery failed", { err: String(err) })); - } - - /** Re-prime every advisor's transcript view (compaction/shake/rewind) without the - * session-level latch reset {@link #resetAdvisorSessionState} performs. */ - #resetAllAdvisorRuntimes(): void { - for (const a of this.#advisors) a.runtime.reset(); - } - - #stopAdvisorRuntime(): void { - // Detach each recorder feed BEFORE aborting its advisor agent: dispose() aborts - // the loop, and an abort emits a final `message_end` we must not enqueue against - // a closing recorder (it would reopen and resurrect an already-released file). - const closes: Promise[] = []; - for (const a of this.#advisors) { - a.agentUnsubscribe?.(); - a.agentUnsubscribe = undefined; - a.runtime.dispose(); - // Capture each close so dispose()/`/drop` can await the queued open+append+close — - // the last advisor turn would otherwise be lost on a fast process exit. - a.recorderClosed = a.recorder.close(); - closes.push(a.recorderClosed); - } - this.#advisorRecorderClosed = Promise.all(closes).then(() => {}); - this.#advisors = []; - this.#advisorYieldQueueUnsubscribe?.(); - this.#advisorYieldQueueUnsubscribe = undefined; - } - - /** Subscribe the advisor agent's finalized messages into the transcript recorder. - * Idempotent-by-replacement: callers detach the prior feed first. Kept separate - * so the re-prime path can mute the feed across an abort-driven reset. */ - #attachAdvisorRecorderFeed(advisor: ActiveAdvisor): void { - advisor.agentUnsubscribe = advisor.agent.subscribe(event => { - if (event.type === "message_end") advisor.recorder.record(event.message); - }); - } - - /** Switch one advisor model while preserving its context and effort invariants. */ - #setAdvisorModel(advisor: ActiveAdvisor, model: Model, requestedThinkingLevel: ThinkingLevel): ThinkingLevel { - const resolvedThinkingLevel = resolveThinkingLevelForModel(model, requestedThinkingLevel); - const nextThinkingLevel = resolvedThinkingLevel ?? ThinkingLevel.Inherit; - advisor.agent.setModel(model); - advisor.agent.setThinkingLevel(toReasoningEffort(nextThinkingLevel)); - advisor.agent.setDisableReasoning(shouldDisableReasoning(nextThinkingLevel)); - advisor.agent.appendOnlyContext?.invalidateForModelChange(); - advisor.model = model; - advisor.thinkingLevel = nextThinkingLevel; - return nextThinkingLevel; - } - - /** Restore an advisor's configured primary once its fallback cooldown expires. */ - async #maybeRestoreAdvisorRetryFallbackPrimary(advisor: ActiveAdvisor): Promise { - const fallback = advisor.retryFallback; - if (!fallback || this.#getRetryFallbackRevertPolicy() !== "cooldown-expiry") return; - - const originalSelector = parseRetryFallbackSelector(fallback.originalSelector, this.#modelRegistry); - if (!originalSelector) { - advisor.retryFallback = undefined; - advisor.retryFallbackPendingSuccess = false; - return; - } - const currentSelector = formatRetryFallbackSelector(advisor.agent.state.model, advisor.thinkingLevel); - if (currentSelector === originalSelector.raw) { - if (!this.#isRetryFallbackSelectorSuppressed(originalSelector)) { - advisor.retryFallback = undefined; - advisor.retryFallbackPendingSuccess = false; - } - return; - } - if (this.#isRetryFallbackSelectorSuppressed(originalSelector)) return; - - const resolvedPrimary = resolveModelOverride([originalSelector.raw], this.#modelRegistry, this.settings); - const primaryModel = - resolvedPrimary.model ?? this.#modelRegistry.find(originalSelector.provider, originalSelector.id); - if (!primaryModel) return; - const apiKey = await this.#modelRegistry.getApiKey(primaryModel, advisor.providerSessionId); - if (!apiKey) return; - - const thinkingToApply = - advisor.thinkingLevel === fallback.lastAppliedThinkingLevel - ? fallback.originalThinkingLevel - : advisor.thinkingLevel; - this.#setAdvisorModel(advisor, primaryModel, thinkingToApply); - this.settings.getStorage()?.recordModelUsage(formatModelStringWithRouting(primaryModel)); - advisor.retryFallback = undefined; - advisor.retryFallbackPendingSuccess = false; - } - - /** - * Apply the advisor's configured provider-failure fallback chain after - * same-provider credential rotation has no usable sibling. - */ - async #recoverAdvisorTurn( - advisor: ActiveAdvisor, - error: unknown, - failedMessages: readonly AgentMessage[], - ): Promise { - if (error instanceof AdvisorOutputQuarantinedError) return false; - - const failedMessage = failedMessages.findLast( - (message): message is AssistantMessage => message.role === "assistant", - ); - if (failedMessage?.stopReason !== "error") { - // Stream setup can reject before any assistant turn is recorded (e.g. - // an HTTP 429 thrown from prompt()); classify the raw error so a - // structural usage limit still marks the exhausted credential. - const message = error instanceof Error ? error.message : String(error); - if (!AIError.isUsageLimit(error) && !isUsageLimitOutcome(extractHttpStatusFromError(error), message)) { - return false; - } - const currentModel = advisor.agent.state.model; - const outcome = await this.#modelRegistry.authStorage.markUsageLimitReached( - currentModel.provider, - advisor.providerSessionId, - { - retryAfterMs: extractRetryHint(undefined, message), - baseUrl: currentModel.baseUrl, - modelId: currentModel.id, - }, - ); - return outcome.switched; - } - if (failedMessage.content.some(block => block.type === "toolCall")) return false; - - const currentModel = advisor.agent.state.model; - const message = failedMessage.errorMessage ?? (error instanceof Error ? error.message : String(error)); - const errorId = AIError.classifyMessage({ - api: currentModel.api, - errorId: failedMessage.errorId, - errorMessage: message, - errorStatus: failedMessage.errorStatus, - }); - if (AIError.is(errorId, AIError.Flag.Abort) || AIError.is(errorId, AIError.Flag.UserInterrupt)) return false; - if (AIError.isContextOverflow(failedMessage, currentModel.contextWindow ?? 0)) return false; - - const currentSelector = formatRetryFallbackSelector(currentModel, advisor.thinkingLevel); - - const retryAfterMs = extractRetryHint(undefined, message); - if ( - AIError.is(errorId, AIError.Flag.UsageLimit) || - isUsageLimitOutcome(extractHttpStatusFromError(error), message) - ) { - const outcome = await this.#modelRegistry.authStorage.markUsageLimitReached( - currentModel.provider, - advisor.providerSessionId, - { - retryAfterMs, - baseUrl: currentModel.baseUrl, - modelId: currentModel.id, - }, - ); - if (outcome.switched) return true; - } - - const retrySettings = this.settings.getGroup("retry"); - if (!retrySettings.enabled || !retrySettings.modelFallback) return false; - const role = advisor.retryFallback?.role ?? this.#resolveRetryFallbackRole(currentSelector, currentModel); - if (!role || this.#findRetryFallbackCandidates(role, currentSelector, currentModel).length === 0) return false; - - this.#noteRetryFallbackCooldown(currentSelector, retryAfterMs, message); - for (const selector of this.#findRetryFallbackCandidates(role, currentSelector, currentModel)) { - if (this.#isRetryFallbackSelectorSuppressed(selector)) continue; - const resolved = resolveModelOverride([selector.raw], this.#modelRegistry, this.settings); - const candidate = resolved.model ?? this.#modelRegistry.find(selector.provider, selector.id); - if (!candidate || modelsAreEqual(candidate, currentModel)) continue; - const apiKey = await this.#modelRegistry.getApiKey(candidate, advisor.providerSessionId); - if (!apiKey) continue; - - const originalThinkingLevel = advisor.thinkingLevel; - const requestedThinkingLevel = selector.thinkingLevel ?? originalThinkingLevel; - const nextThinkingLevel = this.#setAdvisorModel(advisor, candidate, requestedThinkingLevel); - if (advisor.retryFallback) { - advisor.retryFallback.lastAppliedThinkingLevel = nextThinkingLevel; - } else { - advisor.retryFallback = { - role, - originalSelector: currentSelector, - originalThinkingLevel, - lastAppliedThinkingLevel: nextThinkingLevel, - }; - } - advisor.retryFallbackPendingSuccess = true; - this.settings.getStorage()?.recordModelUsage(formatModelStringWithRouting(candidate)); - await this.#emitSessionEvent({ - type: "retry_fallback_applied", - from: currentSelector, - to: selector.raw, - role, - }); - return true; - } - return false; - } - - async #promoteAdvisorContextModel(advisor: ActiveAdvisor, currentModel: Model): Promise { - const promotionSettings = this.settings.getGroup("contextPromotion"); - if (!promotionSettings.enabled) return false; - const contextWindow = currentModel.contextWindow ?? 0; - if (contextWindow <= 0) return false; - const targetModel = await this.#resolveContextPromotionTarget(currentModel, contextWindow); - if (!targetModel) return false; - - // Preserve this advisor's own thinking level (a configured `model:...:high` - // keeps its suffix across a promotion); only the model changes. - const advisorThinkingLevel = advisor.thinkingLevel; - try { - this.#setAdvisorModel(advisor, targetModel, advisorThinkingLevel); - logger.debug("Advisor context promotion switched model on overflow", { - advisor: advisor.name, - from: `${currentModel.provider}/${currentModel.id}`, - to: `${targetModel.provider}/${targetModel.id}`, - }); - return true; - } catch (error) { - logger.warn("Advisor context promotion failed", { - advisor: advisor.name, - from: `${currentModel.provider}/${currentModel.id}`, - to: `${targetModel.provider}/${targetModel.id}`, - error: String(error), - }); - return false; - } - } - - async #maintainAdvisorContext(advisor: ActiveAdvisor, incomingTokens: number): Promise { - await this.#maybeRestoreAdvisorRetryFallbackPrimary(advisor); - const agent = advisor.agent; - - const compactionSettings = this.settings.getGroup("compaction"); - if (compactionSettings.strategy === "off") return false; - if (!compactionSettings.enabled) return false; - - const advisorModel = agent.state.model; - const contextWindow = advisorModel.contextWindow ?? 0; - if (contextWindow <= 0) return false; - - const messages = agent.state.messages; - const estimateOptions = { excludeEncryptedReasoning: true } as const; - let storedConversationTokens = 0; - for (const message of messages) { - storedConversationTokens += estimateTokens(message, estimateOptions); - } - // Provider usage (including cache reads and generated output) is the - // trustworthy anchor for accumulated context. Add only the trailing incoming - // delta to that arm. Floor it by a full local estimate — fixed advisor system - // prompt, tool schemas, stored messages, and incoming delta — so provider - // under-reporting or payload transforms cannot suppress maintenance. - const providerContextTokens = this.#estimateAdvisorContextTokens(messages) + incomingTokens; - const localContextTokens = - countTokens(agent.state.systemPrompt) + - estimateToolSchemaTokens(agent.state.tools) + - storedConversationTokens + - incomingTokens; - const contextTokens = compactionContextTokens(providerContextTokens, localContextTokens); - - if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) { - return false; - } - - // 1. Try promotion first - if (await this.#promoteAdvisorContextModel(advisor, advisorModel)) { - // Promotion succeeded, check if new model has enough space - const newModel = agent.state.model; - const newWindow = newModel.contextWindow ?? 0; - if (newWindow > 0) { - const stillNeedsCompaction = shouldCompact(contextTokens, newWindow, compactionSettings); - if (!stillNeedsCompaction) return false; - } - } - - // 2. Run compaction on advisor messages - const pathEntries: SessionEntry[] = messages.map((message, i) => { - const id = `msg-${i}`; - const parentId = i > 0 ? `msg-${i - 1}` : null; - const timestamp = String(message.timestamp || Date.now()); - - if (message.role === "compactionSummary") { - const advisorSummary = message as AdvisorCompactionSummaryMessage; - return { - type: "compaction", - id, - parentId, - timestamp, - summary: message.summary, - shortSummary: message.shortSummary, - firstKeptEntryId: advisorSummary.firstKeptEntryId || `msg-${i + 1}`, - tokensBefore: message.tokensBefore, - } satisfies CompactionEntry; - } - - return { - type: "message", - id, - parentId, - timestamp, - message, - } satisfies SessionMessageEntry; - }); - - const availableModels = this.#modelRegistry.getAvailable(); - const candidates = this.#resolveCompactionModelCandidates(advisorModel, availableModels); - if (candidates.length === 0) { - // No compaction candidates, fallback to re-prime - return true; - } - const advisorProviderSessionId = getOrCreateAdvisorProviderSessionId( - this.#advisorProviderSessionIds, - this.sessionId, - advisor.slug, - ); - const preparation = prepareCompaction(pathEntries, compactionSettings, advisorModel); - if (!preparation) { - // Cannot prepare compaction, fallback to re-prime - return true; - } - - const advisorCompactionThinkingLevel: ThinkingLevel | undefined = agent.state.disableReasoning - ? ThinkingLevel.Off - : agent.state.thinkingLevel; - - // Advisor state is in-memory-only, so snapcompact's frame archive has no - // stable SessionEntry preserveData slot to carry across future advisor - // maintenance runs. Use an LLM summary even when the primary session is - // configured for snapcompact. - - let compactResult: CompactionResult | undefined; - let lastError: unknown; - // Instrument the advisor's overflow-compaction one-shot like the primary - // compaction path so the advisor model's maintenance call also emits spans. - const telemetry = resolveTelemetry(agent.telemetry, advisorProviderSessionId); - - const codexCompaction = createCodexCompactionContext({ - trigger: "auto", - reason: "context_limit", - phase: "pre_turn", - }); - - for (const candidate of candidates) { - const apiKey = await this.#modelRegistry.getApiKey(candidate, advisorProviderSessionId); - if (!apiKey) continue; - - try { - compactResult = await compact( - preparation, - candidate, - this.#modelRegistry.resolver(candidate, advisorProviderSessionId), - undefined, - undefined, - { - thinkingLevel: advisorCompactionThinkingLevel, - convertToLlm: messages => this.#convertToLlmForSideRequest(messages), - telemetry, - tools: agent.state.tools, - sessionId: advisorProviderSessionId, - promptCacheKey: advisorProviderSessionId, - providerSessionState: this.#providerSessionState, - codexCompaction, - }, - ); - break; - } catch (error) { - lastError = error; - } - } - - if (!compactResult) { - logger.warn("Advisor compaction failed, falling back to re-prime", { error: String(lastError) }); - return true; - } - - const summary = compactResult.summary; - const shortSummary = compactResult.shortSummary; - const firstKeptEntryId = compactResult.firstKeptEntryId; - const tokensBefore = compactResult.tokensBefore; - - // The retained messages still carry provider usage from before this - // compaction. Record their exact array boundary on the in-memory summary so - // only assistants appended afterward can become the next usage anchor. - const advisorUsageAnchorStartIndex = preparation.recentMessages.length + 1; - const summaryMessage = { - ...createCompactionSummaryMessage(summary, tokensBefore, new Date().toISOString(), shortSummary), - firstKeptEntryId, - advisorUsageAnchorStartIndex, - } satisfies AdvisorCompactionSummaryMessage; - - agent.replaceMessages([summaryMessage, ...preparation.recentMessages]); - return false; - } - /** Model registry for API key resolution and model discovery */ get modelRegistry(): ModelRegistry { return this.#modelRegistry; @@ -4095,7 +1571,7 @@ export class AgentSession { /** TTSR manager for time-traveling stream rules */ get ttsrManager(): TtsrManager | undefined { - return this.#ttsrManager; + return this.#ttsr.manager; } /** Secret obfuscator, when secrets are configured; /share redaction reuses it. */ @@ -4105,7 +1581,7 @@ export class AgentSession { /** Whether a TTSR abort is pending (stream was aborted to inject rules) */ get isTtsrAbortPending(): boolean { - return this.#ttsrAbortPending; + return this.#ttsr.abortPending; } /** Whether an expected internal plan-mode abort is pending. Consumed by @@ -4449,8 +1925,7 @@ export class AgentSession { if (event.type !== "agent_end") { const processing = this.#processAgentEvent(event); if ((event.type === "message_start" || event.type === "message_end") && isAdvisorCard(event.message)) { - this.#pendingAdvisorCardEvents.add(processing); - void processing.finally(() => this.#pendingAdvisorCardEvents.delete(processing)).catch(() => {}); + this.#advisors.trackCardEvent(processing); } return processing; } @@ -4604,12 +2079,12 @@ export class AgentSession { if (this.#sessionMessageAlreadyPersisted(message)) return; if (message.role === "assistant") { const assistantMsg = message as AssistantMessage; - if (this.#isClassifierRefusal(assistantMsg)) return; + if (this.#recovery.isClassifierRefusal(assistantMsg)) return; if (isEmptyErrorTurn(assistantMsg)) return; if (assistantMsg.stopReason !== "aborted" && assistantMsg.stopReason !== "error" && assistantMsg.usage) { assistantMsg.contextSnapshot = { promptTokens: calculatePromptTokens(assistantMsg.usage), - nonMessageTokens: this.#pendingContextSnapshot?.nonMessageTokens ?? computeNonMessageTokens(this), + nonMessageTokens: this.#stats.pendingNonMessageTokens ?? computeNonMessageTokens(this), }; } } @@ -4711,17 +2186,7 @@ export class AgentSession { // and only successful mutating tools tick — read-only exploration is // not progress an agent could mark done. if (event.type === "message_end" && event.message.role === "toolResult") { - const { toolName, isError } = event.message; - if (toolName === "todo") { - this.#mutationsSinceLastTodoTouch = 0; - } else if (!isError && MID_RUN_TODO_NUDGE_MUTATING_TOOLS[toolName]) { - this.#mutationsSinceLastTodoTouch++; - } - // A tool actually ran. Clear the post-reminder suppression synchronously - // too: the settle check (`#checkTodoCompletion` in agent_end maintenance) - // can otherwise read the stale flag when a tool result and the terminal - // stop land in the same tick, swallowing the earned re-escalation. - this.#todoReminderAwaitingProgress = false; + this.#todo.onToolResult(event.message.toolName, event.message.isError); } // Track the settled assistant turn synchronously as well: agent_end // maintenance reads `#lastAssistantMessage`, and when a turn's events all @@ -4810,15 +2275,11 @@ export class AgentSession { } if (event.type === "turn_start") { - this.#resetStreamingEditState(); - // TTSR: Reset buffer on turn start - this.#ttsrManager?.resetBuffer(); + this.#streamingEditGuard.reset(); + this.#ttsr.onTurnStart(); } - // TTSR: Increment message count on turn end (for repeat-after-gap tracking) - if (event.type === "turn_end" && this.#ttsrManager) { - this.#ttsrManager.incrementMessageCount(); - } + if (event.type === "turn_end") this.#ttsr.onTurnEnd(); // Finalize the tool-choice queue's in-flight yield after tools have executed. // This must happen at turn_end (not message_end) because onInvoked handlers // run during tool execution, which happens between message_end and turn_end. @@ -4846,38 +2307,7 @@ export class AgentSession { } } - // TTSR: Check for pattern matches on assistant text/thinking and tool argument deltas - if (event.type === "message_update" && this.#ttsrManager?.hasRules()) { - const assistantEvent = event.assistantMessageEvent; - let matchContext: TtsrMatchContext | undefined; - let streamingToolCall: ToolCall | undefined; - - if (assistantEvent.type === "text_delta") { - matchContext = { source: "text" }; - } else if (assistantEvent.type === "thinking_delta") { - matchContext = { source: "thinking" }; - } else if (assistantEvent.type === "toolcall_delta") { - streamingToolCall = this.#getStreamingToolCallBlock(event.message, assistantEvent.contentIndex); - matchContext = this.#getTtsrToolMatchContext(streamingToolCall, assistantEvent.contentIndex); - } - - if (matchContext && "delta" in assistantEvent) { - const targetMessageTimestamp = event.message.role === "assistant" ? event.message.timestamp : undefined; - const matches = this.#checkTtsrStream(assistantEvent.delta, matchContext, streamingToolCall); - if (matches.length > 0 && this.#handleTtsrMatches(matches, matchContext, targetMessageTimestamp)) { - return; - } - // ast-grep `astCondition` rules match against the reconstructed edit/write - // snapshot, which only exists for tool argument streams. The native worker - // call is async, so this path is awaited and self-throttled by the manager. - if (matchContext.source === "tool" && this.#ttsrManager?.hasAstRules()) { - const astMatches = await this.#checkTtsrAstStream(matchContext, streamingToolCall); - if (astMatches.length > 0 && this.#handleTtsrMatches(astMatches, matchContext, targetMessageTimestamp)) { - return; - } - } - } - } + if (await this.#ttsr.checkMessageUpdate(event)) return; if ( event.type === "message_update" && @@ -4885,14 +2315,14 @@ export class AgentSession { event.assistantMessageEvent.type === "toolcall_delta" || event.assistantMessageEvent.type === "toolcall_end") ) { - void this.#preCacheStreamingEditFile(event); + this.#streamingEditGuard.preCache(event); } if ( event.type === "message_update" && (event.assistantMessageEvent.type === "toolcall_end" || event.assistantMessageEvent.type === "toolcall_delta") ) { - this.#maybeAbortStreamingEdit(event); + this.#streamingEditGuard.maybeAbort(event); } // Handle session persistence @@ -4909,7 +2339,7 @@ export class AgentSession { event.message.attribution ?? "agent", ); if (event.message.role === "custom" && event.message.customType === "ttsr-injection") { - this.#markTtsrInjected(this.#extractTtsrRuleNames(event.message.details)); + this.#ttsr.markInjectedFromDetails(event.message.details); } } else { this.#persistSessionMessageIfMissing(event.message); @@ -4945,7 +2375,7 @@ export class AgentSession { } if ( assistantMsg.disabledFeatures?.includes("priority") && - this.#serviceTierByFamily.anthropic === "priority" + this.serviceTierByFamily.anthropic === "priority" ) { this.setServiceTierFamily("anthropic", undefined); this.emitNotice( @@ -4954,40 +2384,11 @@ export class AgentSession { "priority", ); } - // Resolve TTSR resume gate before checking for new deferred injections. - // Gate on #ttsrAbortPending, not stopReason: a non-TTSR abort (e.g. streaming - // edit) also produces stopReason === "aborted" but has no continuation coming. - // Only skip when #ttsrAbortPending is true (TTSR continuation is imminent). - if (!this.#ttsrAbortPending) { - this.#resolveTtsrResume(); - } - this.#queueDeferredTtsrInjectionIfNeeded(assistantMsg); - if (this.#handoffAbortController) { - this.#skipPostTurnMaintenanceAssistantTimestamp = assistantMsg.timestamp; - } - if ( - assistantMsg.stopReason !== "error" && - assistantMsg.stopReason !== "aborted" && - !this.#isEmptyAssistantStop(assistantMsg) && - this.#retryAttempt > 0 - ) { - if (this.#activeRetryFallback && this.model) { - await this.#emitSessionEvent({ - type: "retry_fallback_succeeded", - model: formatRetryFallbackSelector(this.model, this.thinkingLevel), - role: this.#activeRetryFallback.role, - }); - } - const recoveredErrors = await this.#markPendingRecoveredRetryErrors(assistantMsg); - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: true, - attempt: this.#retryAttempt, - recoveredErrors, - }); - this.#clearPendingRecoveredRetryErrors(); - this.#retryAttempt = 0; + this.#ttsr.onAssistantMessageEnd(assistantMsg); + if (this.#handoff.isGeneratingHandoff) { + this.#maintenance.skipPostTurnMaintenanceAssistantTimestamp = assistantMsg.timestamp; } + await this.#recovery.onAssistantSettledSuccessfully(assistantMsg); if (assistantMsg.provider === "opencode-go") { this.#modelRegistry.authStorage.recordUsageCost(assistantMsg.provider, assistantMsg.usage.cost.total, { sessionId: this.#activeProviderSessionId(), @@ -5002,17 +2403,12 @@ export class AgentSession { const semanticResult = semanticToolResult(toolName, event.message); const semanticDetails = isRecord(semanticResult?.details) ? semanticResult.details : undefined; // Invalidate streaming edit cache when edit tool completes to prevent stale data - const editedPath = details ? getStringProperty(details, "path") : undefined; + const editedPath = details ? stringProperty(details, "path") : undefined; if (toolName === "edit" && editedPath) { - this.#invalidateFileCacheForPath(editedPath); + this.#streamingEditGuard.invalidate(editedPath); } - // TodoTool commits its state during execute. Replaying the result after - // awaited event fan-out can overwrite a newer call from the same batch. - const phases = details?.phases; - if (toolName === "todo" && !isError && details && Array.isArray(phases) && phases.every(isTodoPhase)) { - if (this.#isTodoInitResult(details, toolCallId)) { - this.#scheduleReplanTitleRefresh(); - } + if (toolName === "todo" && !isError && details && this.#todo.onTodoResultDetails(details, toolCallId)) { + this.#scheduleReplanTitleRefresh(); } if (toolName === "todo" && isError) { const errorText = content.find(part => part.type === "text")?.text; @@ -5061,7 +2457,7 @@ export class AgentSession { const activeMessages = this.agent.state.messages; // TTSR retry work runs concurrently and clears the live flag before // maintenance can emit agent_end, so preserve the state at settle entry. - const ttsrAbortPendingAtAgentEnd = this.#ttsrAbortPending; + const ttsrAbortPendingAtAgentEnd = this.#ttsr.abortPending; const emitAgentEndNotification = async (options?: { willContinue?: boolean }) => { // Public agent_end is held out of the eager display pass and emitted // here after maintenance routing, tagged isTerminal so subscribers can @@ -5134,8 +2530,8 @@ export class AgentSession { await this.#modelRegistry.authStorage.remove("github-copilot"); } - if (this.#skipPostTurnMaintenanceAssistantTimestamp === msg.timestamp) { - this.#skipPostTurnMaintenanceAssistantTimestamp = undefined; + if (this.#maintenance.skipPostTurnMaintenanceAssistantTimestamp === msg.timestamp) { + this.#maintenance.skipPostTurnMaintenanceAssistantTimestamp = undefined; this.#lastSuccessfulYieldToolCallId = undefined; maintenanceRoute("skip-post-turn-maintenance"); await emitAgentEndNotification(); @@ -5157,7 +2553,7 @@ export class AgentSession { ? "successful-yield-active-goal-checkCompaction" : "post-yield-trailing-stop-active-goal-checkCompaction", ); - const compactionTask = this.#checkCompaction(successfulYieldMessage); + const compactionTask = this.#maintenance.checkCompaction(successfulYieldMessage); this.#trackPostPromptTask(compactionTask); await compactionTask; } else if (successfulYieldMessage) { @@ -5177,7 +2573,7 @@ export class AgentSession { // tool_result and corrupts message history. The handler also // schedules its own retry, so a real empty stop never needs the // active-goal threshold pre-empt below. - if (await this.#handleEmptyAssistantStop(msg)) { + if (await this.#recovery.handleEmptyAssistantStop(msg)) { maintenanceRoute("empty-stop-handled"); await emitAgentEndNotification({ willContinue: true }); return; @@ -5187,7 +2583,7 @@ export class AgentSession { let checkedCompaction = false; if (activeGoal) { maintenanceRoute("active-goal-pre-empt-checkCompaction"); - const compactionTask = this.#checkCompaction(msg); + const compactionTask = this.#maintenance.checkCompaction(msg); this.#trackPostPromptTask(compactionTask); compactionResult = await compactionTask; checkedCompaction = true; @@ -5198,7 +2594,7 @@ export class AgentSession { continuationScheduled: compactionResult.continuationScheduled, automaticContinuationBlocked: compactionResult.automaticContinuationBlocked === true, }); - this.#resolveRetry(); + this.#recovery.resolveRetry(); await emitAgentEndNotification( compactionResult.continuationScheduled ? { willContinue: true } : undefined, ); @@ -5206,14 +2602,14 @@ export class AgentSession { } } - if (await this.#handleUnexpectedAssistantStop(msg)) { + if (await this.#recovery.handleUnexpectedAssistantStop(msg)) { maintenanceRoute("unexpected-stop-handled"); await emitAgentEndNotification({ willContinue: true }); return; } - if (this.#isRetryableReasonlessAbort(msg)) { - const didRetry = await this.#handleRetryableError(msg, { allowModelFallback: false }); + if (this.#recovery.isRetryableReasonlessAbort(msg)) { + const didRetry = await this.#recovery.handleRetryableError(msg, { allowModelFallback: false }); if (didRetry) { await emitAgentEndNotification({ willContinue: true }); return; @@ -5224,7 +2620,7 @@ export class AgentSession { // continuations — except TTSR self-repair, which already scheduled a // hidden retry while #ttsrAbortPending is still true. if (msg.stopReason === "aborted") { - this.#resolveRetry(); + this.#recovery.resolveRetry(); this.#resetSessionStopContinuationState(); await emitAgentEndNotification(ttsrAbortPendingAtAgentEnd ? { willContinue: true } : undefined); return; @@ -5232,16 +2628,16 @@ export class AgentSession { // Fireworks Fast variants degrade to their base model on a failed turn — // including hard router errors the generic retry classifier rejects — so // run this gate before the standard retryability check. - if (this.#isFireworksFastFallbackEligible(msg)) { - const didRetry = await this.#handleRetryableError(msg, { fireworksFastFallback: true }); + if (this.#recovery.isFireworksFastFallbackEligible(msg)) { + const didRetry = await this.#recovery.handleRetryableError(msg, { fireworksFastFallback: true }); if (didRetry) { await emitAgentEndNotification({ willContinue: true }); return; } } - const resumeResolvedStreamStall = this.#canResumeResolvedStreamStall(msg); - if (resumeResolvedStreamStall || this.#isRetryableError(msg)) { - const didRetry = await this.#handleRetryableError( + const resumeResolvedStreamStall = this.#recovery.canResumeResolvedStreamStall(msg); + if (resumeResolvedStreamStall || this.#recovery.isRetryableError(msg)) { + const didRetry = await this.#recovery.handleRetryableError( msg, resumeResolvedStreamStall ? { preserveFailedTurn: true } : undefined, ); @@ -5249,13 +2645,13 @@ export class AgentSession { await emitAgentEndNotification({ willContinue: true }); return; } - } else if (this.#isHardErrorFallbackEligible(msg)) { + } else if (this.#recovery.isHardErrorFallbackEligible(msg)) { // A non-retryable hard error on a model covered by a configured // fallback chain: retrying the SAME model is pointless, but a // DIFFERENT model is a fresh chance — consult the chain before // surfacing the failure. #handleRetryableError bails out (no // backoff-retry of the failing model) when no switch happens. - const didRetry = await this.#handleRetryableError(msg, { hardErrorFallback: true }); + const didRetry = await this.#recovery.handleRetryableError(msg, { hardErrorFallback: true }); if (didRetry) { await emitAgentEndNotification({ willContinue: true }); return; @@ -5269,9 +2665,9 @@ export class AgentSession { // Fall through to the standard error tail so `session_stop` hooks (block, // continue, telemetry) still fire — matching the pre-fix flow for // `stopReason === "error"`. - if (this.#isClassifierRefusal(msg)) { + if (this.#recovery.isClassifierRefusal(msg)) { this.#prunedTerminalRefusal = msg; - this.#removeAssistantMessageFromActiveContext(msg); + this.#recovery.removeAssistantMessageFromActiveContext(msg); } else if (!AIError.isContextOverflow(msg, this.model?.contextWindow ?? 0)) { // No retry, fallback, or compaction continuation fired: this errored // turn ends the run. #persistSessionMessageIfMissing dropped it as an @@ -5280,27 +2676,17 @@ export class AgentSession { // Idempotent and a no-op for non-empty turns. Content-less overflow // rejections stay live-UI only per the auto-compaction progress guard: // persisting one would replay an empty assistant turn on reload. - await this.#persistTerminalEmptyErrorTurn(msg); + await this.#recovery.persistTerminalEmptyErrorTurn(msg); } - this.#resolveRetry(); + this.#recovery.resolveRetry(); if (!checkedCompaction) { maintenanceRoute("bottom-checkCompaction"); - const compactionTask = this.#checkCompaction(msg); + const compactionTask = this.#maintenance.checkCompaction(msg); this.#trackPostPromptTask(compactionTask); compactionResult = await compactionTask; } - if (msg.stopReason === "error" && this.#retryAttempt > 0 && !compactionResult.continuationScheduled) { - const attempt = this.#retryAttempt; - this.#retryAttempt = 0; - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt, - finalError: msg.errorMessage, - }); - this.#clearPendingRecoveredRetryErrors(); - } + await this.#recovery.onErrorSettledWithoutRetry(msg, compactionResult); // Stop-time todo reconciliation only fires at a text-only final stop. A run // that ends still mid-tool-use (deadline hit, context full, etc.) skips the // reminder so we don't pile a follow-up onto an already in-flight turn. @@ -5333,7 +2719,7 @@ export class AgentSession { await emitAgentEndNotification({ willContinue: true }); return; } - const todoContinuationScheduled = await this.#checkTodoCompletion(msg); + const todoContinuationScheduled = await this.#todo.checkCompletion(msg); if (todoContinuationScheduled) { await emitAgentEndNotification({ willContinue: true }); return; @@ -5353,31 +2739,6 @@ export class AgentSession { } }; - /** Resolve the pending retry promise */ - #resolveRetry(): void { - if (this.#retryResolve) { - this.#retryResolve(); - this.#retryResolve = undefined; - this.#retryPromise = undefined; - } - } - - /** Create the TTSR resume gate promise if one doesn't already exist. */ - #ensureTtsrResumePromise(): void { - if (this.#ttsrResumePromise) return; - const { promise, resolve } = Promise.withResolvers(); - this.#ttsrResumePromise = promise; - this.#ttsrResumeResolve = resolve; - } - - /** Resolve and clear the TTSR resume gate. */ - #resolveTtsrResume(): void { - if (!this.#ttsrResumeResolve) return; - this.#ttsrResumeResolve(); - this.#ttsrResumeResolve = undefined; - this.#ttsrResumePromise = undefined; - } - #ensurePostPromptTasksPromise(): void { if (this.#postPromptTasksPromise) return; const { promise, resolve } = Promise.withResolvers(); @@ -5455,7 +2816,7 @@ export class AgentSession { } this.#beginInFlight(); try { - await this.#maybeRestoreRetryFallbackPrimary(); + await this.#recovery.maybeRestoreRetryFallbackPrimary(); if (signal.aborted || this.#isDisposed) { this.#skipAgentContinue("post-restore-unavailable", options); return; @@ -5506,7 +2867,7 @@ export class AgentSession { // delegate-via-tasks / phased-todo reminders on this auto-resumed turn. This runs // at invocation (past the abort check below), so an aborted continuation queues // nothing; scoped to this request via prependMessages, never the shared queue. - const eagerNudges = this.#buildPostCompactionEagerNudges(); + const eagerNudges = this.#todo.buildPostCompactionEagerNudges(); await this.#promptWithMessage( { role: "developer", @@ -5542,7 +2903,7 @@ export class AgentSession { async #cancelPostPromptTasks(): Promise { this.#postPromptTasksAbortController.abort(); this.#postPromptTasksAbortController = new AbortController(); - this.#resolveTtsrResume(); + this.#ttsr.resolveResume(); const pendingTasks = Array.from(this.#postPromptTasks); if (pendingTasks.length === 0) { @@ -5568,12 +2929,14 @@ export class AgentSession { // its promise must resolve on the abort, not block on a queued // steer/follow-up that the post-abort drain starts as a fresh turn. if (generation !== undefined && this.#promptGeneration !== generation) return; - if (this.#retryPromise) { - await this.#retryPromise; + const retryPromise = this.#recovery.retryPromise; + if (retryPromise) { + await retryPromise; continue; } - if (this.#ttsrResumePromise) { - await this.#ttsrResumePromise; + const ttsrResumeGate = this.#ttsr.resumeGate; + if (ttsrResumeGate) { + await ttsrResumeGate; continue; } if (this.#postPromptTasksPromise) { @@ -5591,95 +2954,6 @@ export class AgentSession { } } - #formatTtsrAbortReason(rules: Rule[]): string { - const label = rules.length === 1 ? "rule" : "rules"; - const ruleNames = rules.map(rule => rule.name).join(", "); - return `TTSR matched ${label}: ${ruleNames}`; - } - - /** Get TTSR injection payload and clear pending injections. */ - #getTtsrInjectionContent(): { content: string; rules: Rule[] } | undefined { - if (this.#pendingTtsrInjections.length === 0) return undefined; - const rules = this.#pendingTtsrInjections; - const content = rules - .map(r => - prompt.render(ttsrInterruptTemplate, { - name: r.name, - path: this.#displayRulePath(r.path), - content: r.content, - }), - ) - .join("\n\n"); - this.#pendingTtsrInjections = []; - return { content, rules }; - } - - /** - * Render a rule's file path for model-facing TTSR injections without leaking - * the absolute home directory: cwd-relative when the rule lives in the - * project, `~`-relative when it lives under home, else the raw path. - */ - #displayRulePath(rulePath: string): string { - const cwdRel = - relativePathWithinRoot(this.sessionManager.getCwd(), rulePath) ?? - this.#displayPathWithinRoot(this.sessionManager.getCwd(), rulePath); - if (cwdRel) return cwdRel; - const homeRel = relativePathWithinRoot(os.homedir(), rulePath); - if (homeRel) return `~/${homeRel}`; - return rulePath; - } - - #displayPathWithinRoot(root: string, candidate: string): string | null { - const relative = path.relative(path.resolve(root), path.resolve(candidate)); - return relative && !relative.startsWith("..") && !path.isAbsolute(relative) ? relative : null; - } - - #addPendingTtsrInjections(rules: Rule[]): void { - const seen = new Set(this.#pendingTtsrInjections.map(rule => rule.name)); - for (const rule of rules) { - if (seen.has(rule.name)) continue; - this.#pendingTtsrInjections.push(rule); - seen.add(rule.name); - } - } - - /** Tool-call id whose argument deltas triggered a TTSR match, when known. */ - #extractTtsrToolCallId(matchContext: TtsrMatchContext): string | undefined { - if (matchContext.source !== "tool") return undefined; - const key = matchContext.streamKey; - if (typeof key !== "string" || !key.startsWith("toolcall:")) return undefined; - const id = key.slice("toolcall:".length); - return id.length > 0 ? id : undefined; - } - - #addPerToolTtsrInjections(toolCallId: string, rules: Rule[]): void { - const bucket = this.#perToolTtsrInjections.get(toolCallId) ?? []; - const seen = new Set(bucket.map(rule => rule.name)); - // Dedupe against rules already bucketed for other tool calls in this - // same assistant message so one rule attaches to exactly one tool call. - const claimedElsewhere = new Set(); - for (const [otherId, otherBucket] of this.#perToolTtsrInjections) { - if (otherId === toolCallId) continue; - for (const rule of otherBucket) claimedElsewhere.add(rule.name); - } - const newlyAdded: string[] = []; - for (const rule of rules) { - if (seen.has(rule.name) || claimedElsewhere.has(rule.name)) continue; - bucket.push(rule); - seen.add(rule.name); - newlyAdded.push(rule.name); - } - if (bucket.length === 0) return; - this.#perToolTtsrInjections.set(toolCallId, bucket); - // Claim the rules in the TTSR manager so subsequent deltas in this same - // turn (e.g. a sibling tool call's argument stream) don't re-match them. - // Persistence still happens in #ttsrAfterToolCall when the tool actually - // produces a result we can fold the reminder into. - if (newlyAdded.length > 0) { - this.#ttsrManager?.markInjectedByNames(newlyAdded); - } - } - #afterToolCall(ctx: AfterToolCallContext): AfterToolCallResult | undefined { if ( this.#isTerminalYieldToolResult({ @@ -5692,440 +2966,9 @@ export class AgentSession { this.#synchronouslyTerminatedYieldToolCallIds.add(ctx.toolCall.id); this.agent.abort(TERMINAL_TOOL_RESULT_ABORT_REASON); } - return this.#ttsrAfterToolCall(ctx); + return this.#ttsr.afterToolCall(ctx); } - /** `afterToolCall` hook: fold any per-tool TTSR reminders into the result. */ - #ttsrAfterToolCall(ctx: AfterToolCallContext): AfterToolCallResult | undefined { - const rules = this.#perToolTtsrInjections.get(ctx.toolCall.id); - if (!rules || rules.length === 0) return undefined; - this.#perToolTtsrInjections.delete(ctx.toolCall.id); - const reminder = rules - .map(r => - prompt.render(ttsrToolReminderTemplate, { - name: r.name, - path: this.#displayRulePath(r.path), - content: r.content, - }), - ) - .join("\n\n"); - // The TTSR manager was already claimed at bucket time; only persistence remains. - const ruleNames = rules.map(r => r.name.trim()).filter(n => n.length > 0); - if (ruleNames.length > 0) { - this.sessionManager.appendTtsrInjection(ruleNames); - } - return { - content: [{ type: "text", text: reminder }, ...ctx.result.content], - }; - } - - #extractTtsrRuleNames(details: unknown): string[] { - if (!details || typeof details !== "object" || Array.isArray(details)) { - return []; - } - const rules = (details as { rules?: unknown }).rules; - if (!Array.isArray(rules)) { - return []; - } - return rules.filter((ruleName): ruleName is string => typeof ruleName === "string"); - } - - #markTtsrInjected(ruleNames: string[]): void { - const uniqueRuleNames = Array.from( - new Set(ruleNames.map(ruleName => ruleName.trim()).filter(ruleName => ruleName.length > 0)), - ); - if (uniqueRuleNames.length === 0) { - return; - } - this.#ttsrManager?.markInjectedByNames(uniqueRuleNames); - this.sessionManager.appendTtsrInjection(uniqueRuleNames); - } - - #findTtsrAssistantIndex(targetTimestamp: number | undefined): number { - const messages = this.agent.state.messages; - for (let i = messages.length - 1; i >= 0; i--) { - const message = messages[i]; - if (message.role !== "assistant") { - continue; - } - if (targetTimestamp === undefined || message.timestamp === targetTimestamp) { - return i; - } - } - return -1; - } - - #shouldInterruptForTtsrMatch(matches: Rule[], matchContext: TtsrMatchContext): boolean { - const globalMode = this.#ttsrManager?.getSettings().interruptMode ?? "always"; - for (const rule of matches) { - const mode = rule.interruptMode ?? globalMode; - if (mode === "never") continue; - if (mode === "prose-only" && (matchContext.source === "text" || matchContext.source === "thinking")) - return true; - if (mode === "tool-only" && matchContext.source === "tool") return true; - if (mode === "always") return true; - } - return false; - } - - #queueDeferredTtsrInjectionIfNeeded(assistantMsg: AssistantMessage): void { - if (assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error") { - // Tools that hadn't started by abort/error will never produce results to - // fold injections into — drop their stale per-tool entries. - this.#perToolTtsrInjections.clear(); - } - if (this.#ttsrAbortPending || this.#pendingTtsrInjections.length === 0) { - return; - } - if (assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error") { - this.#pendingTtsrInjections = []; - return; - } - - const injection = this.#getTtsrInjectionContent(); - if (!injection) { - return; - } - this.agent.followUp({ - role: "custom", - customType: "ttsr-injection", - content: injection.content, - display: false, - details: { rules: injection.rules.map(rule => rule.name) }, - attribution: "agent", - timestamp: Date.now(), - }); - this.#ensureTtsrResumePromise(); - // Mark as injected after this custom message is delivered and persisted (handled in message_end). - // followUp() only enqueues; resume on the next tick once streaming settles. - this.#scheduleAgentContinue({ - delayMs: 1, - generation: this.#promptGeneration, - onSkip: () => { - this.#resolveTtsrResume(); - }, - shouldContinue: () => { - if (this.agent.state.isStreaming || !this.agent.hasQueuedMessages()) { - this.#resolveTtsrResume(); - return false; - } - return true; - }, - onError: () => { - this.#resolveTtsrResume(); - }, - }); - } - - /** Extract the tool-call block a toolcall_delta event refers to, if present. */ - #getStreamingToolCallBlock(message: AgentMessage, contentIndex: number): ToolCall | undefined { - if (message.role !== "assistant") { - return undefined; - } - - const content = message.content; - if (!Array.isArray(content) || contentIndex < 0 || contentIndex >= content.length) { - return undefined; - } - - const block = content[contentIndex]; - if (!block || typeof block !== "object" || block.type !== "toolCall") { - return undefined; - } - - return block as ToolCall; - } - - /** Build TTSR match context for tool call argument deltas. */ - #getTtsrToolMatchContext(toolCall: ToolCall | undefined, contentIndex: number): TtsrMatchContext { - const context: TtsrMatchContext = { source: "tool" }; - if (!toolCall) { - return context; - } - - context.toolName = toolCall.name; - context.streamKey = toolCall.id ? `toolcall:${toolCall.id}` : `tool:${toolCall.name}:${contentIndex}`; - context.filePaths = this.#extractTtsrToolFilePaths(toolCall); - return context; - } - - /** - * Resolve the file paths a tool call would touch for TTSR path-glob matching. - * - * Prefer the tool's own `matcherPaths` hook — it understands the wire format - * (hashline `[path#TAG]` section headers, apply_patch envelope markers) and - * surfaces paths the generic top-level argument scan never sees. Fall back - * to {@link #extractTtsrFilePathsFromArgs} for tools that pass paths as - * `path`/`paths` arguments and for tool calls whose payload has not yet - * streamed a header. - */ - #extractTtsrToolFilePaths(toolCall: ToolCall): string[] | undefined { - const args = toolCall.arguments ?? {}; - const tools = this.agent.state.tools; - const tool = - tools.find(t => t.name === toolCall.name) ?? - tools.find(t => t.customWireName !== undefined && t.customWireName === toolCall.name); - const toolPaths = tool?.matcherPaths?.(args); - if (toolPaths && toolPaths.length > 0) { - const normalized = toolPaths.flatMap(p => this.#normalizeTtsrPathCandidates(p)); - if (normalized.length > 0) return Array.from(new Set(normalized)); - } - return this.#extractTtsrFilePathsFromArgs(args); - } - - /** - * Match a stream delta against TTSR rules. - * - * Tool argument streams prefer the tool's `matcherDigest` normalization — the - * real content the call introduces — over the raw argument delta, so rule - * conditions written against source text keep working regardless of the - * tool's wire format (hashline patches, JSON-escaped strings, ...). - */ - #checkTtsrStream(delta: string, matchContext: TtsrMatchContext, toolCall: ToolCall | undefined): Rule[] { - const manager = this.#ttsrManager; - if (!manager) { - return []; - } - const entries = this.#resolveTtsrMatcherEntries(toolCall); - if (entries) { - const matches: Rule[] = []; - for (const entry of entries) { - matches.push(...manager.checkSnapshot(entry.digest, this.#perFileTtsrContext(matchContext, entry.path))); - } - return matches; - } - const digest = this.#resolveTtsrMatcherDigest(toolCall); - if (digest !== undefined) { - return manager.checkSnapshot(digest, matchContext); - } - return manager.checkDelta(delta, matchContext); - } - - /** Reconstruct the tool's normalized source snapshot via its `matcherDigest`, if any. */ - #resolveTtsrMatcherDigest(toolCall: ToolCall | undefined): string | undefined { - const tool = this.#resolveTtsrTool(toolCall); - return tool?.matcherDigest?.(toolCall?.arguments ?? {}); - } - - /** - * Per-file split of a streamed call (one entry per touched file paired with - * the digest of only that file's added lines). Lets {@link #checkTtsrStream} - * and {@link #checkTtsrAstStream} evaluate each file in isolation so a - * path-scoped rule like `tool:edit(*.ts)` never fires on text that belongs - * to a sibling Markdown hunk in a multi-file payload. - */ - #resolveTtsrMatcherEntries(toolCall: ToolCall | undefined): readonly { path: string; digest: string }[] | undefined { - const tool = this.#resolveTtsrTool(toolCall); - const entries = tool?.matcherEntries?.(toolCall?.arguments ?? {}); - return entries && entries.length > 0 ? entries : undefined; - } - - #resolveTtsrTool(toolCall: ToolCall | undefined) { - if (!toolCall) return undefined; - const tools = this.agent.state.tools; - return ( - tools.find(t => t.name === toolCall.name) ?? - tools.find(t => t.customWireName !== undefined && t.customWireName === toolCall.name) - ); - } - - /** - * Replace `matchContext`'s `filePaths` + `streamKey` so a per-file entry - * gets its own glob-eligible path and its own TTSR buffer/repeat tracking - * (each file's stream is independent inside the same tool call). - */ - #perFileTtsrContext(base: TtsrMatchContext, filePath: string): TtsrMatchContext { - const filePaths = this.#normalizeTtsrPathCandidates(filePath); - return { - ...base, - filePaths: filePaths.length > 0 ? filePaths : [filePath], - streamKey: base.streamKey ? `${base.streamKey}#${filePath}` : undefined, - }; - } - - /** - * Match ast-grep `astCondition` rules against the reconstructed tool snapshot. - * - * Only edit/write tool streams expose a `matcherDigest`, which is the real source - * the call introduces; AST matching needs that (and a language inferred from the - * path argument), so non-digest streams never produce AST matches. - */ - async #checkTtsrAstStream(matchContext: TtsrMatchContext, toolCall: ToolCall | undefined): Promise { - const manager = this.#ttsrManager; - if (!manager) { - return []; - } - const entries = this.#resolveTtsrMatcherEntries(toolCall); - if (entries) { - const matches: Rule[] = []; - for (const entry of entries) { - matches.push( - ...(await manager.checkAstSnapshot(entry.digest, this.#perFileTtsrContext(matchContext, entry.path))), - ); - } - return matches; - } - const digest = this.#resolveTtsrMatcherDigest(toolCall); - if (digest === undefined) { - return []; - } - return manager.checkAstSnapshot(digest, matchContext); - } - - /** - * Route TTSR matches to either a per-tool injection or a stream-interrupting - * retry. Returns true when the stream was aborted and the caller should stop - * processing this event. - */ - #handleTtsrMatches( - matches: Rule[], - matchContext: TtsrMatchContext, - targetMessageTimestamp: number | undefined, - ): boolean { - // Decide first: a non-interrupting tool-source match attaches to the - // specific tool call's result instead of driving a loop-wide follow-up. - const shouldInterrupt = this.#shouldInterruptForTtsrMatch(matches, matchContext); - const matchedToolId = this.#extractTtsrToolCallId(matchContext); - const perToolId = shouldInterrupt ? undefined : matchedToolId; - if (perToolId) { - this.#addPerToolTtsrInjections(perToolId, matches); - this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); - return false; - } - - // Queue rules for injection; mark as injected only after successful enqueue. - this.#addPendingTtsrInjections(matches); - if (!shouldInterrupt) { - return false; - } - - // Abort the stream immediately — do not gate on extension callbacks - this.#ttsrAbortPending = true; - this.#ensureTtsrResumePromise(); - const abortReason = this.#formatTtsrAbortReason(matches); - this.agent.abort( - matchedToolId - ? createToolScopedAbortReason( - abortReason, - { [matchedToolId]: abortReason }, - "TTSR interrupt on another tool call", - ) - : abortReason, - ); - // Notify extensions (fire-and-forget, does not block abort) - this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); - // Schedule retry after a short delay - const retryToken = ++this.#ttsrRetryToken; - const generation = this.#promptGeneration; - this.#schedulePostPromptTask( - async () => { - if (this.#ttsrRetryToken !== retryToken) { - this.#resolveTtsrResume(); - return; - } - - const targetAssistantIndex = this.#findTtsrAssistantIndex(targetMessageTimestamp); - if (!this.#ttsrAbortPending || this.#promptGeneration !== generation || targetAssistantIndex === -1) { - this.#ttsrAbortPending = false; - this.#pendingTtsrInjections = []; - this.#perToolTtsrInjections.clear(); - this.#resolveTtsrResume(); - return; - } - this.#ttsrAbortPending = false; - this.#perToolTtsrInjections.clear(); - const ttsrSettings = this.#ttsrManager?.getSettings(); - if (ttsrSettings?.contextMode === "discard") { - // Remove the partial/aborted assistant turn from agent state - this.agent.replaceMessages(this.agent.state.messages.slice(0, targetAssistantIndex)); - } - // Inject TTSR rules as system reminder before retry - const injection = this.#getTtsrInjectionContent(); - if (injection) { - const details = { rules: injection.rules.map(rule => rule.name) }; - this.agent.appendMessage({ - role: "custom", - customType: "ttsr-injection", - content: injection.content, - display: false, - details, - attribution: "agent", - timestamp: Date.now(), - }); - this.sessionManager.appendCustomMessageEntry( - "ttsr-injection", - injection.content, - false, - details, - "agent", - ); - this.#markTtsrInjected(details.rules); - } - try { - await this.agent.continue(); - } catch { - this.#resolveTtsrResume(); - } - }, - { delayMs: 50 }, - ); - return true; - } - - /** Extract path-like arguments from tool call payload for TTSR glob matching. */ - #extractTtsrFilePathsFromArgs(args: unknown): string[] | undefined { - if (!args || typeof args !== "object" || Array.isArray(args)) { - return undefined; - } - - const rawPaths: string[] = []; - for (const [key, value] of Object.entries(args)) { - const normalizedKey = key.toLowerCase(); - if (typeof value === "string" && (normalizedKey === "path" || normalizedKey.endsWith("path"))) { - rawPaths.push(value); - continue; - } - if (Array.isArray(value) && (normalizedKey === "paths" || normalizedKey.endsWith("paths"))) { - for (const candidate of value) { - if (typeof candidate === "string") { - rawPaths.push(candidate); - } - } - } - } - - const normalizedPaths = rawPaths.flatMap(pathValue => this.#normalizeTtsrPathCandidates(pathValue)); - if (normalizedPaths.length === 0) { - return undefined; - } - - return Array.from(new Set(normalizedPaths)); - } - - /** Convert a path argument into stable relative/absolute candidates for glob checks. */ - #normalizeTtsrPathCandidates(rawPath: string): string[] { - const trimmed = rawPath.trim(); - if (trimmed.length === 0) { - return []; - } - - const normalizedInput = trimmed.replaceAll("\\", "/"); - const candidates = new Set([normalizedInput]); - if (normalizedInput.startsWith("./")) { - candidates.add(normalizedInput.slice(2)); - } - - const cwd = this.sessionManager.getCwd(); - const absolutePath = path.isAbsolute(trimmed) ? path.normalize(trimmed) : path.resolve(cwd, trimmed); - candidates.add(absolutePath.replaceAll("\\", "/")); - - const relativePath = path.relative(cwd, absolutePath).replaceAll("\\", "/"); - if (relativePath && relativePath !== "." && !relativePath.startsWith("../") && relativePath !== "..") { - candidates.add(relativePath); - } - - return Array.from(candidates); - } /** Find the last assistant message in agent state (including aborted ones) */ #findLastAssistantMessage(): AssistantMessage | undefined { const messages = this.agent.state.messages; @@ -6138,305 +2981,6 @@ export class AgentSession { return undefined; } - #resetStreamingEditState(): void { - this.#streamingEditAbortTriggered = false; - this.#streamingEditCheckedLineCounts.clear(); - this.#streamingEditPrecheckedToolCallIds.clear(); - this.#streamingEditFileCache.clear(); - } - - #activeToolCallLoopGuard(): ToolCallLoopGuard | undefined { - if (this.settings.get("model.toolCallLoopGuard.enabled") !== true) { - this.#toolCallLoopGuard = undefined; - this.#toolCallLoopGuardSettingsKey = undefined; - return undefined; - } - - const threshold = this.settings.get("model.toolCallLoopGuard.threshold"); - const exemptTools = this.settings - .get("model.toolCallLoopGuard.exemptTools") - .filter((tool): tool is string => typeof tool === "string" && tool.length > 0); - const settingsKey = `${threshold}:${JSON.stringify(exemptTools)}`; - if (!this.#toolCallLoopGuard || this.#toolCallLoopGuardSettingsKey !== settingsKey) { - this.#toolCallLoopGuard = new ToolCallLoopGuard({ threshold, exemptTools }); - this.#toolCallLoopGuardSettingsKey = settingsKey; - } - return this.#toolCallLoopGuard; - } - - #maybeInjectToolCallLoopRedirect(messages: AgentMessage[], detection: RepeatedToolCallDetection): void { - const content = prompt.render(toolCallLoopRedirectTemplate, { - tool_name: detection.toolName, - count: detection.count, - arguments_summary: detection.argumentsSummary, - result_summary: detection.resultSummary || "(no text result)", - }); - const details = { - toolName: detection.toolName, - count: detection.count, - argumentsSummary: detection.argumentsSummary, - resultSummary: detection.resultSummary, - }; - logger.warn("cross-turn tool-call loop detected", { - toolName: detection.toolName, - count: detection.count, - }); - const redirectMessage: CustomMessage = { - role: "custom", - customType: TOOL_CALL_LOOP_REDIRECT_TYPE, - content, - display: false, - details, - attribution: "agent", - timestamp: Date.now(), - }; - messages.push(redirectMessage); - if (this.agent.state.messages !== messages) { - this.agent.appendMessage(redirectMessage); - } - this.sessionManager.appendCustomMessageEntry(TOOL_CALL_LOOP_REDIRECT_TYPE, content, false, details, "agent"); - } - - /** - * Whether the Gemini header-runaway guard applies to the current model: the loop - * guard is on (settings + `PI_NO_THINKING_LOOP_GUARD`), the tool-call reminder is - * enabled, and the active model is a Gemini thinking model. - */ - #geminiHeaderGuardActive(): boolean { - const model = this.model; - return ( - process.env.PI_NO_THINKING_LOOP_GUARD !== "1" && - this.settings.get("model.loopGuard.enabled") === true && - this.settings.get("model.loopGuard.toolCallReminder") === true && - model !== undefined && - isGeminiThinkingModel(model) - ); - } - - /** - * Feed streamed assistant events to the Gemini header-runaway detector. Each - * reasoning block (`thinking_start`) re-arms a fresh detector when the guard - * applies; thinking deltas accumulate thought-summary headers; assistant prose - * or a tool call ends the run. On the threshold hit, interrupts the stream (see - * {@link #interruptGeminiHeaderRunaway}). Runs synchronously inside the - * assistant-message interceptor so the abort lands before more budget burns. - * Armed on `thinking_start` (not `turn_start`, which the agent loop skips for the - * first turn) so the very first reasoning block is guarded too. - */ - #maybeInterruptGeminiHeaderRunaway(message: AssistantMessage, event: AssistantMessageEvent): void { - if (event.type === "thinking_start") { - this.#geminiHeaderDetector = this.#geminiHeaderGuardActive() ? new GeminiHeaderRunDetector() : undefined; - return; - } - const detector = this.#geminiHeaderDetector; - if (!detector) return; - if (event.type === "thinking_delta") { - if (detector.push(event.delta)) this.#interruptGeminiHeaderRunaway(detector.count, message.timestamp); - return; - } - // Leaving the reasoning channel ends the run: the consecutive-header count - // only matters within one uninterrupted stretch of reasoning. - if (event.type === "text_start" || event.type === "toolcall_start") { - detector.reset(); - } - } - - /** - * Interrupt a Gemini reasoning stream that has emitted too many consecutive - * planning headers without calling a tool. Aborts the live turn, discards the - * stalled reasoning-only turn (so its partial, loop-fueling thinking is neither - * replayed nor reloaded), injects a hidden tool-call reminder, and continues. - * `targetTimestamp` identifies the turn being aborted so the post-prompt task - * can drop exactly it. - */ - #interruptGeminiHeaderRunaway(headerCount: number, targetTimestamp: number): void { - logger.warn("Gemini reasoning-header runaway; interrupting to require a tool call", { - model: this.model?.id, - provider: this.model?.provider, - headers: headerCount, - }); - this.emitNotice( - "warning", - `Interrupted ${headerCount} planning headers with no tool call; reminded the model to issue one.`, - "loop-guard", - ); - this.agent.abort(GEMINI_HEADER_INTERRUPT_REASON); - const generation = this.#promptGeneration; - this.#schedulePostPromptTask(async signal => { - if (signal.aborted || this.#isDisposed || this.#promptGeneration !== generation) return; - // Let the aborted stream finish unwinding so continue() doesn't race it. - await this.agent.waitForIdle(); - if (signal.aborted || this.#isDisposed || this.#promptGeneration !== generation) return; - const aborted = this.agent.state.messages.findLast( - (m): m is AssistantMessage => m.role === "assistant" && m.timestamp === targetTimestamp, - ); - if (aborted) this.#discardAssistantTurn(aborted); - const content = prompt.render(geminiToolReminderTemplate, { count: headerCount }); - const details = { headers: headerCount }; - this.agent.appendMessage({ - role: "custom", - customType: GEMINI_TOOL_REMINDER_TYPE, - content, - display: false, - details, - attribution: "agent", - timestamp: Date.now(), - }); - this.sessionManager.appendCustomMessageEntry(GEMINI_TOOL_REMINDER_TYPE, content, false, details, "agent"); - try { - await this.agent.continue(); - } catch (err) { - logger.warn("gemini tool-call reminder continue failed", { error: String(err) }); - } - }); - } - - #getStreamingEditToolCall(event: AgentEvent): - | { - toolCall: ToolCall; - path: string; - resolvedPath: string; - diff?: string; - op?: string; - rename?: string; - } - | undefined { - if (event.type !== "message_update") return undefined; - if (event.message.role !== "assistant") return undefined; - - const contentIndex = event.assistantMessageEvent.contentIndex ?? 0; - const messageContent = event.message.content; - if (!Array.isArray(messageContent) || contentIndex < 0 || contentIndex >= messageContent.length) { - return undefined; - } - - const toolCall = messageContent[contentIndex] as ToolCall; - if (toolCall.name !== "edit") return undefined; - - const args = toolCall.arguments; - if (!args || typeof args !== "object" || Array.isArray(args)) return undefined; - if ("old_text" in args || "new_text" in args) return undefined; - - const path = typeof args.path === "string" ? args.path : undefined; - if (!path) return undefined; - - // `local://` URLs (e.g. local://PLAN.md for plan-mode) resolve to a real - // on-disk artifacts path; pre-caching works as long as we ask the - // local-protocol handler. Other internal-scheme URLs have no local - // filesystem representation; skip pre-cache entirely for those — the - // edit tool itself will reject them through its normal dispatch path. - const resolvedPath = this.#resolveSessionFsPath(path); - if (resolvedPath === undefined) return undefined; - - return { - toolCall, - path, - resolvedPath, - diff: typeof args.diff === "string" ? args.diff : undefined, - op: typeof args.op === "string" ? args.op : undefined, - rename: typeof args.rename === "string" ? args.rename : undefined, - }; - } - - #lastStreamingEditToolCallId: string | undefined; - #abortStreamingEditForAutoGeneratedPath(toolCall: ToolCall, path: string, resolvedPath: string): void { - if (this.#lastStreamingEditToolCallId === toolCall.id) return; - this.#lastStreamingEditToolCallId = toolCall.id; - void assertEditableFile(resolvedPath, path).catch(err => { - // peekFile and other I/O can reject with ENOENT, etc. Only ToolError means - // auto-generated detection; other failures are left for the edit tool. - if (!(err instanceof ToolError)) return; - if (this.#lastStreamingEditToolCallId !== toolCall.id) return; - - if (!this.#streamingEditAbortTriggered) { - this.#streamingEditAbortTriggered = true; - logger.warn("Streaming edit aborted due to auto-generated file guard", { - toolCallId: toolCall.id, - path, - }); - this.agent.abort(); - } - }); - } - - #preCacheStreamingEditFile(event: AgentEvent): void { - if (this.#streamingEditAbortTriggered) return; - if (event.type !== "message_update") return; - - const assistantEvent = event.assistantMessageEvent; - if ( - assistantEvent.type !== "toolcall_start" && - assistantEvent.type !== "toolcall_delta" && - assistantEvent.type !== "toolcall_end" - ) { - return; - } - - const streamingEdit = this.#getStreamingEditToolCall(event); - if (!streamingEdit) return; - - // The auto-generated guard runs unconditionally: editing a generated file - // is never the user's intent, and the cost of a false-positive abort is one - // wasted turn vs. silently corrupting a regenerated source. - const shouldCheckAutoGenerated = - !streamingEdit.toolCall.id || !this.#streamingEditPrecheckedToolCallIds.has(streamingEdit.toolCall.id); - if (shouldCheckAutoGenerated) { - if (streamingEdit.toolCall.id) { - this.#streamingEditPrecheckedToolCallIds.add(streamingEdit.toolCall.id); - } - this.#abortStreamingEditForAutoGeneratedPath( - streamingEdit.toolCall, - streamingEdit.path, - streamingEdit.resolvedPath, - ); - } - - // File-cache priming feeds #maybeAbortStreamingEdit's removed-lines check, - // which is the optional patch-preview verification gated by - // edit.streamingAbort. Skip the read when the setting is off. - if (this.settings.get("edit.streamingAbort")) { - this.#ensureFileCache(streamingEdit.resolvedPath); - } - } - - #ensureFileCache(resolvedPath: string): void { - if (this.#streamingEditFileCache.has(resolvedPath)) return; - - try { - const rawText = fs.readFileSync(resolvedPath, "utf-8"); - const { text } = stripBom(rawText); - this.#streamingEditFileCache.set(resolvedPath, normalizeToLF(text)); - } catch { - // Don't cache on read errors (including ENOENT) - let the edit tool handle them - } - } - - /** Invalidate cache for a file after an edit completes to prevent stale data */ - #invalidateFileCacheForPath(filePath: string): void { - const resolvedPath = this.#resolveSessionFsPath(filePath); - if (resolvedPath === undefined) return; - this.#streamingEditFileCache.delete(resolvedPath); - } - - /** - * Resolve a path supplied to a tool to a real filesystem path. - * - * - `local://` URLs route through the local-protocol handler so they map - * onto the session's on-disk artifacts directory; pre-caching, ENOENT - * handling, and post-edit invalidation all work normally. - * - Other internal-scheme URLs have no local filesystem path; this returns - * `undefined` so callers skip filesystem-only operations. - * - Cwd-relative and absolute paths resolve via `resolveToCwd`. - */ - #resolveSessionFsPath(filePath: string): string | undefined { - const normalized = normalizeLocalScheme(filePath); - if (normalized.startsWith("local:")) { - return resolveLocalUrlToPath(normalized, this.#localProtocolOptions()); - } - if (isInternalUrlPath(normalized)) return undefined; - return resolveToCwd(normalized, this.sessionManager.getCwd()); - } - #localProtocolOptions(): LocalProtocolOptions { return { getArtifactsDir: () => this.sessionManager.getArtifactsDir(), @@ -6444,129 +2988,6 @@ export class AgentSession { }; } - #maybeAbortStreamingEdit(event: AgentEvent): void { - if (!this.settings.get("edit.streamingAbort")) return; - if (this.#streamingEditAbortTriggered) return; - if (event.type !== "message_update") return; - - const assistantEvent = event.assistantMessageEvent; - if (assistantEvent.type !== "toolcall_end" && assistantEvent.type !== "toolcall_delta") return; - - const streamingEdit = this.#getStreamingEditToolCall(event); - if (!streamingEdit?.toolCall.id) return; - - const { toolCall, path, resolvedPath, diff, op, rename } = streamingEdit; - if (!diff) return; - if (op && op !== "update") return; - - if (!diff.includes("\n")) return; - const lastNewlineIndex = diff.lastIndexOf("\n"); - if (lastNewlineIndex < 0) return; - const diffForCheck = diff.endsWith("\n") ? diff : diff.slice(0, lastNewlineIndex + 1); - if (diffForCheck.trim().length === 0) return; - - let normalizedDiff = normalizeDiff(diffForCheck.replace(/\r/g, "")); - if (!normalizedDiff) return; - // Deobfuscate the diff so removed lines match real file content - if (this.#obfuscator) normalizedDiff = this.#obfuscator.deobfuscate(normalizedDiff); - if (!normalizedDiff) return; - const lines = normalizedDiff.split("\n"); - const hasChangeLine = lines.some(line => line.startsWith("+") || line.startsWith("-")); - if (!hasChangeLine) return; - - const lineCount = lines.length; - const lastChecked = this.#streamingEditCheckedLineCounts.get(toolCall.id); - if (lastChecked !== undefined && lineCount <= lastChecked) return; - this.#streamingEditCheckedLineCounts.set(toolCall.id, lineCount); - - const removedLines = lines - .filter(line => line.startsWith("-") && !line.startsWith("--- ")) - .map(line => line.slice(1)); - if (removedLines.length > 0) { - let cachedContent = this.#streamingEditFileCache.get(resolvedPath); - if (cachedContent === undefined) { - this.#ensureFileCache(resolvedPath); - cachedContent = this.#streamingEditFileCache.get(resolvedPath); - } - if (cachedContent !== undefined) { - const missing = removedLines.find(line => !cachedContent.includes(normalizeToLF(line))); - if (missing) { - this.#streamingEditAbortTriggered = true; - logger.warn("Streaming edit aborted due to patch preview failure", { - toolCallId: toolCall.id, - path, - error: `Failed to find expected lines in ${path}:\n${missing}`, - }); - this.agent.abort(); - } - return; - } - if (assistantEvent.type === "toolcall_delta") return; - void this.#checkRemovedLinesAsync(toolCall.id, path, resolvedPath, removedLines); - return; - } - - if (assistantEvent.type === "toolcall_delta") return; - void this.#checkPreviewPatchAsync(toolCall.id, path, rename, normalizedDiff); - } - - async #checkRemovedLinesAsync( - toolCallId: string, - path: string, - resolvedPath: string, - removedLines: string[], - ): Promise { - if (this.#streamingEditAbortTriggered) return; - try { - const { text } = stripBom(await Bun.file(resolvedPath).text()); - const normalizedContent = normalizeToLF(text); - const missing = removedLines.find(line => !normalizedContent.includes(normalizeToLF(line))); - if (missing) { - this.#streamingEditAbortTriggered = true; - logger.warn("Streaming edit aborted due to patch preview failure", { - toolCallId, - path, - error: `Failed to find expected lines in ${path}:\n${missing}`, - }); - this.agent.abort(); - } - } catch (err) { - // Ignore ENOENT (file not found) - let the edit tool handle missing files - // Also ignore other errors during async fallback - if (!isEnoent(err)) { - // Log unexpected errors but don't abort - } - } - } - - async #checkPreviewPatchAsync( - toolCallId: string, - path: string, - rename: string | undefined, - normalizedDiff: string, - ): Promise { - if (this.#streamingEditAbortTriggered) return; - try { - await previewPatch( - { path, op: "update", rename, diff: normalizedDiff }, - { - cwd: this.sessionManager.getCwd(), - allowFuzzy: this.settings.get("edit.fuzzyMatch"), - fuzzyThreshold: this.settings.get("edit.fuzzyThreshold"), - }, - ); - } catch (error) { - if (error instanceof ParseError) return; - this.#streamingEditAbortTriggered = true; - logger.warn("Streaming edit aborted due to patch preview failure", { - toolCallId, - path, - error: error instanceof Error ? error.message : String(error), - }); - this.agent.abort(); - } - } - #resetSessionStopContinuationState(): void { this.#sessionStopContinuationCount = 0; this.#sessionStopHookActive = false; @@ -6885,50 +3306,6 @@ export class AgentSession { ); } - #rekeyHindsightMemoryForCurrentSessionId(): void { - if (this.settings.get("memory.backend") !== "hindsight") return; - const sid = this.agent.sessionId; - if (!sid) return; - this.getHindsightSessionState()?.setSessionId(sid); - } - - #rekeyMnemopiMemoryForCurrentSessionId(): void { - if (this.settings.get("memory.backend") !== "mnemopi") return; - const sid = this.agent.sessionId; - if (!sid) return; - this.getMnemopiSessionState()?.setSessionId(sid); - } - - /** New session file: reset auto-recall / retain-threshold counters for the new transcript. */ - #resetHindsightConversationTrackingIfHindsight(): boolean { - if (this.settings.get("memory.backend") !== "hindsight") return false; - const state = this.getHindsightSessionState(); - if (!state || state.aliasOf) return false; - state.resetConversationTracking(); - return true; - } - - #resetMnemopiConversationTrackingIfMnemopi(): boolean { - if (this.settings.get("memory.backend") !== "mnemopi") return false; - const state = this.getMnemopiSessionState(); - if (!state || state.aliasOf) return false; - state.resetConversationTracking(); - return true; - } - - async #resetMemoryContextForNewTranscript(): Promise { - const hadPromotedMemoryPrompt = this.#baseSystemPromptBeforeMemoryPromotion !== undefined; - const resetHindsight = this.#resetHindsightConversationTrackingIfHindsight(); - const resetMnemopi = this.#resetMnemopiConversationTrackingIfMnemopi(); - if (hadPromotedMemoryPrompt) { - this.#baseSystemPrompt = this.#baseSystemPromptBeforeMemoryPromotion!; - this.agent.setSystemPrompt(this.#baseSystemPrompt); - this.#baseSystemPromptBeforeMemoryPromotion = undefined; - } - if (resetHindsight || resetMnemopi || hadPromotedMemoryPrompt) { - await this.refreshBaseSystemPrompt(); - } - } /** Run one abortable auto-learn capture outside the primary agent loop. */ async runAutolearnCapture(capture: (signal: AbortSignal) => Promise): Promise { if (this.#autolearnCaptureTask || this.#isDisposed) return; @@ -6988,15 +3365,15 @@ export class AgentSession { */ beginDispose(): void { this.#isDisposed = true; - this.cancelLocalMemoryStartup(); + this.#memory.cancelLocalMemoryStartup(); this.#titleGenerationAbortController.abort(); this.#abortAutolearnCapture(); - this.#flushPendingIrcAsides(); + this.#irc.flushPending(); this.yieldQueue.clear(); this.agent.setAsideMessageProvider(undefined); this.agent.hasIrcInterrupts = undefined; - this.#stopAdvisorRuntime(); - this.#evalExecutionDisposing = true; + this.#advisors.stopRuntime(); + this.#eval.beginDispose(); } /** @@ -7037,24 +3414,6 @@ export class AgentSession { } } - async #disposeEvalKernels(): Promise { - const settled = await this.#prepareEvalExecutionsForDispose(); - if (!settled) { - logger.warn("Detaching retained eval-kernel ownership during dispose while eval execution is still active"); - } - - const results = await Promise.allSettled([ - disposeKernelSessionsByOwner(this.#evalKernelOwnerId), - disposeRubyKernelSessionsByOwner(this.#evalKernelOwnerId), - disposeJuliaKernelSessionsByOwner(this.#evalKernelOwnerId), - ]); - const errors: unknown[] = []; - for (const result of results) { - if (result.status === "rejected") errors.push(result.reason); - } - if (errors.length > 0) throw new AggregateError(errors, "Failed to dispose one or more eval kernels"); - } - async #releaseOwnedBrowserTabs(ownerId: string | undefined): Promise { if (!ownerId) return; try { @@ -7119,14 +3478,14 @@ export class AgentSession { logger.warn("Post-prompt tasks still draining at dispose deadline", { error: String(error) }); } await this.#drainAutolearnCapture(); - await this.#memoryBackendTransition; + await this.#memory.transition; const hindsightState = this.getHindsightSessionState(); const mnemopiState = setMnemopiSessionState(this, undefined); - const advisorRecorderClosed = this.#advisorRecorderClosed; + const advisorRecorderClosed = this.#advisors.recorderClosed(); const results = await Promise.allSettled([ this.#disposeOwnedAsyncJobs(), - this.#disposeEvalKernels(), + this.#eval.disposeKernels(), this.#releaseOwnedBrowserTabs(this.sessionManager.getSessionId()), shutdownTinyTitleClient(), this.#disconnectOwnedMcp(), @@ -7185,8 +3544,7 @@ export class AgentSession { this.#closeAllProviderSessions("fresh session"); this.#freshProviderSessionId = Bun.randomUUIDv7(); this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); + this.#memory.rekeyForCurrentSessionId(); this.agent.appendOnlyContext?.invalidateForModelChange(); return { previousSessionId, @@ -7211,35 +3569,32 @@ export class AgentSession { /** Resolved selector while retry routing is using a fallback model. */ get retryFallbackModel(): string | undefined { - const model = this.model; - return this.#activeRetryFallback && model ? formatRetryFallbackSelector(model, this.thinkingLevel) : undefined; + return this.#recovery.retryFallbackModel; } /** Effective thinking level applied to the agent (the resolved level when `auto`). */ get thinkingLevel(): ThinkingLevel | undefined { - return this.#thinkingLevel; + return this.#models.thinkingLevel; } /** The selector the user configured: `auto` when auto mode is active, else the effective level. */ configuredThinkingLevel(): ConfiguredThinkingLevel | undefined { - return this.#autoThinking ? AUTO_THINKING : this.#thinkingLevel; + return this.#models.configuredThinkingLevel(); } /** True when `auto` thinking mode is active. */ get isAutoThinking(): boolean { - return this.#autoThinking; + return this.#models.isAutoThinking; } /** The level `auto` resolved to for the current turn (undefined until classified). */ autoResolvedThinkingLevel(): Effort | undefined { - return this.#autoResolvedLevel; + return this.#models.autoResolvedThinkingLevel; } - #serviceTierByFamily: ServiceTierByFamily = {}; - /** Live per-family service tiers (OpenAI / Anthropic / Google). */ get serviceTierByFamily(): ServiceTierByFamily { - return this.#serviceTierByFamily; + return this.#models.serviceTierByFamily; } /** Whether agent is currently streaming a response */ @@ -7251,9 +3606,10 @@ export class AgentSession { return this.agent.isAborting; } - /** Wait until streaming and deferred recovery work are fully settled. */ + /** Wait until streaming, event persistence, and deferred recovery work are fully settled. */ async waitForIdle(): Promise { await this.agent.waitForIdle(); + await this.#advisors.waitForPendingCardEvents(); await this.#waitForPostPromptRecovery(); } /** @@ -7261,24 +3617,7 @@ export class AgentSession { * caller prints and drains the final primary response. */ prepareForHeadlessAdvisorDrain(): void { - this.#preserveAdvisorAdvice = true; - } - - async #waitForPendingAdvisorCardEvents(timeoutMs: number): Promise { - const deadline = Date.now() + Math.max(0, timeoutMs); - while (this.#pendingAdvisorCardEvents.size > 0) { - const remainingMs = deadline - Date.now(); - if (remainingMs <= 0) return false; - const settled = Promise.allSettled([...this.#pendingAdvisorCardEvents]).then(() => true as const); - const { promise: timedOut, resolve } = Promise.withResolvers(); - const timer = setTimeout(() => resolve(false), remainingMs); - try { - if (!(await Promise.race([settled, timedOut]))) return false; - } finally { - clearTimeout(timer); - } - } - return true; + this.#advisors.prepareForHeadlessAdvisorDrain(); } /** @@ -7286,22 +3625,8 @@ export class AgentSession { * headless caller disposes the session. Returns `false` and logs work disposal * will abandon when the shared deadline expires or an advisor fails. */ - async waitForAdvisorCatchup(timeoutMs: number): Promise { - const deadline = Date.now() + timeoutMs; - const results = await Promise.all(this.#advisors.map(advisor => advisor.runtime.waitForCatchup(timeoutMs, 1))); - const cardEventsCaughtUp = await this.#waitForPendingAdvisorCardEvents(Math.max(0, deadline - Date.now())); - const abandoned = this.#advisors.filter( - (advisor, index) => results[index] === false && advisor.runtime.backlog > 0, - ); - if (abandoned.length > 0 || !cardEventsCaughtUp) { - logger.warn("advisor shutdown drain incomplete; disposal will abandon reviews or cards", { - timeoutMs, - advisors: abandoned.map(advisor => ({ name: advisor.name, backlog: advisor.runtime.backlog })), - pendingAdvisorCards: this.#pendingAdvisorCardEvents.size, - }); - return false; - } - return true; + waitForAdvisorCatchup(timeoutMs: number): Promise { + return this.#advisors.waitForAdvisorCatchup(timeoutMs); } async drainAsyncJobDeliveriesForAcp(options?: { timeoutMs?: number }): Promise { @@ -7337,888 +3662,172 @@ export class AgentSession { /** Current retry attempt (0 if not retrying) */ get retryAttempt(): number { - return this.#retryAttempt; + return this.#recovery.attempt; } - #getActiveNonMCPToolNames(): string[] { - return this.getEnabledToolNames().filter(name => !isMCPToolName(name) && this.#toolRegistry.has(name)); - } - - /** - * Get the names of currently active tools. - * Returns the names of tools currently set on the agent. - */ + /** Names of tools currently exposed at the top level. */ getActiveToolNames(): string[] { - return this.agent.state.tools.map(t => t.name); + return this.#tools.getActiveToolNames(); } - /** - * Enabled tool names: top-level active tools plus discoverable tools mounted - * under `xd://`. Reconcile callers (MCP refresh, discovery, plan/goal mode) - * build their next active set from this so a mount survives an active-set - * change; {@link getActiveToolNames} stays the top-level-only view. - */ + /** Enabled top-level and discoverable tool names. */ getEnabledToolNames(): string[] { - if (this.#mountedXdevToolNames.size === 0) return this.getActiveToolNames(); - return [...this.getActiveToolNames(), ...this.#mountedXdevToolNames]; + return this.#tools.getEnabledToolNames(); } - /** Names of discoverable tools currently mounted under `xd://` (dynamic mounts). */ + /** Names of dynamic tools mounted under `xd://`. */ getMountedXdevToolNames(): string[] { - return [...this.#mountedXdevToolNames]; + return this.#tools.getMountedXdevToolNames(); } /** Whether the edit tool is registered in this session. */ get hasEditTool(): boolean { - return this.#toolRegistry.has("edit"); + return this.#tools.hasEditTool; } - /** - * Get a tool by name from the registry. - */ + /** Looks up a registered tool by name. */ getToolByName(name: string): AgentTool | undefined { - return this.#toolRegistry.get(name); + return this.#tools.getToolByName(name); } - /** True when the current registry entry for `name` came from a built-in factory. */ + /** Whether a registry entry came from a built-in factory. */ hasBuiltInTool(name: string): boolean { - return this.#builtInToolNames.has(name); + return this.#tools.hasBuiltInTool(name); } - /** - * Get all configured tool names (built-in via --tools or default, plus custom tools). - */ + /** Names of every registered tool. */ getAllToolNames(): string[] { - return Array.from(this.#toolRegistry.keys()); + return this.#tools.getAllToolNames(); } - #wrapRuntimeTool(tool: AgentTool): AgentTool { - const wrapped = wrapToolWithMetaNotice(tool); - return this.#extensionRunner ? new ExtensionToolWrapper(wrapped, this.#extensionRunner) : wrapped; + /** Installs and activates the ephemeral vibe tool set. */ + activateVibeTools(baseToolNames: string[]): Promise { + return this.#tools.activateVibeTools(baseToolNames); } - /** - * Registers the ephemeral vibe tools and activates them alongside `baseToolNames`. - * - * @throws When this session cannot create vibe tools or the factory returns duplicate names. - */ - async activateVibeTools(baseToolNames: string[]): Promise { - const createVibeTools = this.#createVibeTools; - if (!createVibeTools) { - throw new Error("Vibe tools are unavailable in this session."); - } - - const tools = createVibeTools(); - const vibeToolNames = tools.map(tool => tool.name); - if (new Set(vibeToolNames).size !== vibeToolNames.length) { - throw new Error("Vibe tool names must be unique."); - } - - for (const tool of tools) { - if (this.#toolRegistry.has(tool.name)) continue; - this.#toolRegistry.set(tool.name, this.#wrapRuntimeTool(tool)); - this.#builtInToolNames.add(tool.name); - this.#installedVibeToolNames.add(tool.name); - } - - await this.#applyActiveToolsByName([...new Set([...baseToolNames, ...vibeToolNames])]); + /** Uninstalls vibe tools and activates the replacement set. */ + deactivateVibeTools(nextToolNames: string[]): Promise { + return this.#tools.deactivateVibeTools(nextToolNames); } - /** Removes tools installed by {@link activateVibeTools} and activates `nextToolNames`. */ - async deactivateVibeTools(nextToolNames: string[]): Promise { - this.#uninstallVibeTools(); - await this.#applyActiveToolsByName(nextToolNames); - } - - /** - * Removes the ephemeral vibe tools while keeping whatever active tool set the - * session currently holds (minus those vibe tools). Unlike - * {@link deactivateVibeTools}, this never restores a caller-held pre-vibe - * snapshot — use it on the session-switch path, where that snapshot belongs to - * the source session and would clobber the freshly loaded target's tools. - */ - async removeVibeToolsPreservingActive(): Promise { - const removed = new Set(this.#installedVibeToolNames); - this.#uninstallVibeTools(); - const nextActive = this.getActiveToolNames().filter(name => !removed.has(name)); - await this.#applyActiveToolsByName(nextActive); - } - - #uninstallVibeTools(): void { - for (const name of this.#installedVibeToolNames) { - this.#toolRegistry.delete(name); - this.#builtInToolNames.delete(name); - } - this.#installedVibeToolNames.clear(); - } - - #getEditModeSession() { - return { - settings: this.settings, - getActiveModelString: () => (this.model ? formatModelString(this.model) : undefined), - } as const; + /** Removes vibe tools without restoring a source-session snapshot. */ + removeVibeToolsPreservingActive(): Promise { + return this.#tools.removeVibeToolsPreservingActive(); } #resolveActiveEditMode(): EditMode { - return resolveEditMode(this.#getEditModeSession()); + return this.#tools.resolveActiveEditMode(); } - /** Cache key for model-dependent prompt content: displayed id or hidden-policy cohort. */ - #currentPromptModelKey(): string | undefined { - const model = this.model ? formatModelString(this.model) : undefined; - if (!model || this.settings.get("includeModelInPrompt")) return model; - return usesCodexTaskPrompt(model) ? "task-policy:gpt-5.6" : "task-policy:default"; - } - - async #syncAfterModelChange(previousEditMode: EditMode): Promise { - const currentEditMode = this.#resolveActiveEditMode(); - const editModeChanged = previousEditMode !== currentEditMode && this.getActiveToolNames().includes("edit"); - // The system prompt selects model-specific policy even when it does not display the model id. - const modelChanged = this.#currentPromptModelKey() !== this.#promptModelKey; - if (editModeChanged || modelChanged) { - await this.refreshBaseSystemPrompt(); - } + #syncAfterModelChange(previousEditMode: EditMode): Promise { + return this.#tools.syncAfterModelChange(previousEditMode); } + /** Enabled MCP tools in their current presentation partition. */ getSelectedMCPToolNames(): string[] { - // Every connected MCP tool is enabled; presentation (top-level vs xd://) is - // decided by loadMode. Return the enabled MCP tools in the current set. - return this.getEnabledToolNames().filter(name => isMCPToolName(name) && this.#toolRegistry.has(name)); + return this.#tools.getSelectedMCPToolNames(); } - /** - * Wrap a tool with a permission-gate proxy when an ACP client is connected. - * Only wraps tools whose name is in PERMISSION_REQUIRED_TOOLS and only when - * the bridge exposes `requestPermission`. No-ops for all other cases. - * - * When the user has explicitly opted into `yolo` / auto-approve behavior (via - * the SDK/CLI `autoApprove` flag or a configured `tools.approvalMode: yolo`), - * skips the gate unless the per-tool policy explicitly requires a prompt or - * deny. The schema default is also `yolo`, so an explicit configuration or - * explicit session flag is required: default-config ACP sessions keep the - * client-side permission gate. - */ - #wrapToolForAcpPermission(tool: T): T { - const bridge = this.#clientBridge; - // Match the capability+method gating pattern used by read/write/bash. - if (!bridge?.capabilities.requestPermission || !bridge.requestPermission) return tool; - if (!PERMISSION_REQUIRED_TOOLS.has(tool.name)) return tool; - // Skip the gate only on explicit yolo opt-in; honour per-tool policies - // that require a prompt or deny (matching the normal approval wrapper). - if (this.#isExplicitAutoApproveMode()) { - const userPolicies = (this.settings.get("tools.approval") ?? {}) as Record; - const toolPolicy = userPolicies[tool.name]; - if (!toolPolicy || toolPolicy === "allow") return tool; - } - return new Proxy(tool, { - get: (target, prop) => { - if (prop !== "execute") return target[prop as keyof T]; - return async ( - toolCallId: string, - args: unknown, - signal: AbortSignal | undefined, - onUpdate: never, - ctx: never, - ) => { - const permissionIntent = getPermissionIntent(target.name, args); - if (!permissionIntent) { - return await target.execute(toolCallId, args as never, signal, onUpdate, ctx); - } - const command = - target.name === "bash" && args && typeof args === "object" && !Array.isArray(args) - ? getStringProperty(args as Record, "command") - : undefined; - const commandContent = command - ? [{ type: "content" as const, content: { type: "text" as const, text: `$ ${command}` } }] - : undefined; - // Short-circuit on persisted decisions. - const persisted = this.#acpPermissionDecisions.get(permissionIntent.cacheKey); - if (persisted === "allow_always") { - return await target.execute(toolCallId, args as never, signal, onUpdate, ctx); - } - if (persisted === "reject_always") { - throw new ToolError(`Tool call rejected by user (preference)`); - } - if (signal?.aborted) { - throw new ToolAbortError("Permission request cancelled"); - } - type PermissionRaceResult = - | { kind: "permission"; outcome: ClientBridgePermissionOutcome } - | { kind: "aborted" }; - const { promise: abortPromise, resolve: resolveAbort } = Promise.withResolvers(); - const onAbort = () => resolveAbort({ kind: "aborted" }); - signal?.addEventListener("abort", onAbort, { once: true }); - let raced: PermissionRaceResult; - try { - const permissionPromise = bridge.requestPermission!( - { - toolCallId, - toolName: target.name, - title: permissionIntent.title, - ...(target.name === "bash" ? { kind: "execute" } : {}), - status: "pending", - rawInput: args, - ...(commandContent ? { content: commandContent } : {}), - locations: extractPermissionLocations( - args, - this.sessionManager.getCwd(), - permissionIntent.paths, - ), - }, - PERMISSION_OPTIONS, - signal, - ).then(outcome => ({ kind: "permission" as const, outcome })); - raced = await Promise.race([permissionPromise, abortPromise]); - } finally { - signal?.removeEventListener("abort", onAbort); - } - if (raced.kind === "aborted" || signal?.aborted) { - throw new ToolAbortError("Permission request cancelled"); - } - const outcome = raced.outcome; - if (outcome.outcome === "cancelled") { - throw new ToolAbortError("Permission request cancelled"); - } - const selectedOption = PERMISSION_OPTIONS_BY_ID.get(outcome.optionId); - if (!selectedOption) { - throw new ToolError(`Tool permission response used unknown option ID: ${outcome.optionId}`); - } - if (selectedOption.kind === "allow_always") { - this.#acpPermissionDecisions.set(permissionIntent.cacheKey, "allow_always"); - } else if (selectedOption.kind === "reject_always") { - this.#acpPermissionDecisions.set(permissionIntent.cacheKey, "reject_always"); - } - if (selectedOption.kind === "reject_once" || selectedOption.kind === "reject_always") { - throw new ToolError(`Tool call rejected by user (${target.name})`); - } - return await target.execute(toolCallId, args as never, signal, onUpdate, ctx); - }; - }, - }) as T; + #applyActiveToolsByName(toolNames: string[]): Promise { + return this.#tools.applyActiveToolsByName(toolNames); } - #isExplicitAutoApproveMode(): boolean { - return ( - this.#autoApprove || - (this.settings.isConfigured("tools.approvalMode") && this.settings.get("tools.approvalMode") === "yolo") - ); - } - - async #applyActiveToolsByName(toolNames: string[]): Promise { - toolNames = normalizeToolNames(toolNames); - const selectedTools = toolNames.flatMap(name => { - const tool = this.#toolRegistry.get(name); - return tool ? [{ name, tool }] : []; - }); - const xdevReadAvailable = this.#builtInToolNames.has("read") && selectedTools.some(({ name }) => name === "read"); - const isPresentationPinned = (name: string): boolean => - this.#presentationPinnedToolNames?.has(name) === true || this.#runtimeSelectedToolNames?.has(name) === true; - const mountCandidates = selectedTools.filter( - ({ name, tool }) => - this.#xdevRegistry !== undefined && - xdevReadAvailable && - !isPresentationPinned(name) && - isMountableUnderXdev(tool), - ); - - let builtInWriteAvailable = this.#builtInToolNames.has("write"); - if (mountCandidates.length > 0 && !builtInWriteAvailable) { - builtInWriteAvailable = (await this.#ensureWriteRegistered?.()) === true; - if (builtInWriteAvailable) this.#builtInToolNames.add("write"); - } - const mountNames = builtInWriteAvailable ? new Set(mountCandidates.map(({ name }) => name)) : new Set(); - const tools: AgentTool[] = []; - const validToolNames: string[] = []; - const mountedTools: AgentTool[] = []; - for (const { name, tool } of selectedTools) { - if (mountNames.has(name)) { - mountedTools.push(this.#wrapToolForAcpPermission(tool)); - } else { - tools.push(this.#wrapToolForAcpPermission(tool)); - validToolNames.push(name); - } - } - - const pinnedWrite = isPresentationPinned("write"); - const activeDeferrableTool = tools.some(tool => tool.deferrable === true); - const transportNeeded = mountedTools.length > 0 || activeDeferrableTool || this.#planModeState?.enabled === true; - if (transportNeeded && !builtInWriteAvailable) { - builtInWriteAvailable = (await this.#ensureWriteRegistered?.()) === true; - if (builtInWriteAvailable) this.#builtInToolNames.add("write"); - } - if (transportNeeded && builtInWriteAvailable) { - const write = this.#toolRegistry.get("write"); - if (write && !validToolNames.includes("write")) { - tools.push(this.#wrapToolForAcpPermission(write)); - validToolNames.push("write"); - } - } else if ( - !pinnedWrite && - (this.#presentationPinnedToolNames !== undefined || this.#runtimeSelectedToolNames !== undefined) - ) { - const writeNameIndex = validToolNames.indexOf("write"); - if (writeNameIndex >= 0 && this.#builtInToolNames.has("write")) validToolNames.splice(writeNameIndex, 1); - const writeToolIndex = tools.findIndex(tool => tool.name === "write" && this.#builtInToolNames.has("write")); - if (writeToolIndex >= 0) tools.splice(writeToolIndex, 1); - } - - const previousMounted = this.#mountedXdevToolNames; - const previousMountedTools = [...previousMounted].flatMap(name => { - const tool = this.#xdevRegistry?.get(name); - return tool ? [tool] : []; - }); - const previousActiveToolNames = this.getActiveToolNames(); - this.#mountedXdevToolNames = new Set(mountedTools.map(tool => tool.name)); - this.#xdevRegistry?.reconcile(mountedTools); - this.#setActiveToolNames?.(validToolNames); - - let rebuiltSystemPrompt: string[] | undefined; - let rebuiltSignature: string | undefined; - try { - if (this.#rebuildSystemPrompt) { - const signature = this.#computeAppliedToolSignature(validToolNames, tools); - if (signature !== this.#lastAppliedToolSignature) { - const built = await this.#rebuildSystemPrompt(validToolNames, this.#toolRegistry); - rebuiltSystemPrompt = built.systemPrompt; - rebuiltSignature = signature; - } - } - } catch (error) { - this.#mountedXdevToolNames = previousMounted; - this.#xdevRegistry?.reconcile(previousMountedTools); - this.#setActiveToolNames?.(previousActiveToolNames); - throw error; - } - - this.#notifyXdevMountDelta(previousMounted); - this.agent.setTools(tools); - if (rebuiltSystemPrompt && rebuiltSignature) { - if (this.#lastAppliedToolSignature !== undefined) this.#clearInheritedProviderPromptCacheKey(); - this.#baseSystemPrompt = rebuiltSystemPrompt; - this.#baseSystemPromptBeforeMemoryPromotion = undefined; - this.agent.setSystemPrompt(this.#baseSystemPrompt); - this.#lastAppliedToolSignature = rebuiltSignature; - this.#promptModelKey = this.#currentPromptModelKey(); - } - } - - /** - * Record a mid-session `xd://` mount delta for the model without rewriting - * the system prompt: the prompt (and its provider cache prefix) stays - * byte-stable across MCP connects and disconnects. The delta is NOT steered - * immediately — a steered notice landing at a run's stop boundary (or while - * the session is idle) forces an unsolicited extra assistant turn — it is - * coalesced into {@link #pendingXdevMountDelta} and rides along with the - * next prompt (docs + schema stay one `read xd://` away). The full - * docs join the system prompt opportunistically on the next unrelated - * rebuild. - */ - #notifyXdevMountDelta(previousMounted: ReadonlySet): void { - const registry = this.#xdevRegistry; - if (!registry) return; - const current = this.#mountedXdevToolNames; - const addedNames = [...current].filter(name => !previousMounted.has(name)); - const removedNames = [...previousMounted].filter(name => !current.has(name)); - if (addedNames.length === 0 && removedNames.length === 0) return; - // Coalesce against the unannounced delta: an unmount cancels a pending - // mount the model never learned about, and a remount cancels a pending - // unmount. - const pending = this.#pendingXdevMountDelta ?? { added: new Set(), removed: new Set() }; - for (const name of addedNames) { - if (!pending.removed.delete(name)) pending.added.add(name); - } - for (const name of removedNames) { - if (!pending.added.delete(name)) pending.removed.add(name); - } - this.#pendingXdevMountDelta = pending.added.size > 0 || pending.removed.size > 0 ? pending : undefined; - if (this.settings.get("startup.quiet")) return; - const parts: string[] = []; - if (addedNames.length > 0) parts.push(`mounted ${addedNames.join(", ")}`); - if (removedNames.length > 0) parts.push(`unmounted ${removedNames.join(", ")}`); - this.emitNotice("info", `xd://: ${parts.join("; ")}`, "xdev"); - } - - /** - * Render and consume the pending xd:// mount delta as a hidden notice, or - * `undefined` when nothing unannounced is queued. Called from the prompt - * paths so the notice rides along with user input instead of forcing its - * own model turn. - */ #takePendingXdevMountNotice(): CustomMessage | undefined { - const pending = this.#pendingXdevMountDelta; - if (!pending) return undefined; - this.#pendingXdevMountDelta = undefined; - const summaries = new Map(this.#xdevRegistry?.entries().map(entry => [entry.name, entry.summary]) ?? []); - const added = [...pending.added].map(name => ({ name, summary: summaries.get(name) ?? "" })); - const removed = [...pending.removed].map(name => ({ name })); - return { - role: "custom", - customType: XDEV_MOUNT_NOTICE_MESSAGE_TYPE, - content: prompt.render(xdevMountNoticePrompt, { added, removed }), - attribution: "agent", - display: false, - timestamp: Date.now(), - }; + return this.#tools.takePendingXdevMountNotice(); } - /** - * Rediscover disk-backed skills and rebuild prompt-facing state without - * recreating the session. Explicit skill snapshots (`--no-skills`, - * SDK-provided `skills`) remain fixed for the lifetime of the session. - */ - async refreshSkills(): Promise { - if (!this.#skillsReloadable) { - return; - } - - resetCapabilities(); - const skillsSettings = this.settings.getGroup("skills"); - const discovered = await loadSkills({ - ...skillsSettings, - cwd: this.sessionManager.getCwd(), - disabledExtensions: this.settings.get("disabledExtensions") ?? [], - }); - this.#skills = discovered.skills; - this.#skillWarnings = discovered.warnings; - this.#skillsSettings = skillsSettings; - - if (this.#agentKind === "main") { - setActiveSkills(this.#skills); - } - await this.refreshBaseSystemPrompt(); - this.#notifyCommandMetadataChanged(); + /** Rediscovers reloadable skills and refreshes prompt metadata. */ + refreshSkills(): Promise { + return this.#tools.refreshSkills(); } - /** - * Set active tools by name. - * Only tools in the registry can be enabled. Unknown tool names are ignored. - * Also rebuilds the system prompt to reflect the new tool set. - * Changes take effect before the next model call. - */ - async setActiveToolsByName(toolNames: string[]): Promise { - const normalized = normalizeToolNames(toolNames); - // Transport-write eligibility keys off the *current* active set: an ordinary - // selection change should not demote `write` unless it is already active. - await this.#applyToolPresentation( - normalized, - this.#mountedXdevToolNames, - this.getActiveToolNames().includes("write"), - ); + /** Selects enabled tools, ignoring names absent from the registry. */ + setActiveToolsByName(toolNames: string[]): Promise { + return this.#tools.setActiveToolsByName(toolNames); } - /** - * Restore an enabled tool set with its exact top-level versus `xd://` partition. - * - * Both inputs are required because {@link setActiveToolsByName} only receives the - * enabled name list and classifies mounts from the *current* `#mountedXdevToolNames`. - * Rollback/restore callers must pass the snapshotted mounted subset so names that - * were top-level stay pinned (`#runtimeSelectedToolNames`) and names that were under - * `xd://` remain mount-eligible, even when the live mount set has drifted. - * - * Names outside `mountedToolNames` are pinned top-level for this application; - * names in the mounted subset remain eligible for xdev mounting. Delegates the - * actual apply through `#applyActiveToolsByName` and restores the prior runtime - * selection if that apply throws. - */ - async setActiveToolPresentation(toolNames: string[], mountedToolNames: string[]): Promise { - const normalized = normalizeToolNames(toolNames); - // Restoration targets a snapshot, so write eligibility comes from the - // *target* set rather than whatever happens to be active mid-rollback. - await this.#applyToolPresentation( - normalized, - new Set(normalizeToolNames(mountedToolNames)), - normalized.includes("write"), - ); + /** Restores an exact top-level versus `xd://` tool partition. */ + setActiveToolPresentation(toolNames: string[], mountedToolNames: string[]): Promise { + return this.#tools.setActiveToolPresentation(toolNames, mountedToolNames); } - /** - * Shared body for {@link setActiveToolsByName} and {@link setActiveToolPresentation}: - * pins non-mounted names as the runtime selection (holding `write` back when it is - * transport-only) and applies the set, rolling the selection back if apply throws. - */ - async #applyToolPresentation( - normalized: string[], - mounted: ReadonlySet, - writeSelected: boolean, - ): Promise { - const transportWriteActive = - writeSelected && - this.#builtInToolNames.has("write") && - this.#presentationPinnedToolNames?.has("write") !== true && - this.#runtimeSelectedToolNames?.has("write") !== true && - (mounted.size > 0 || this.#planModeState?.enabled === true); - const previousRuntimeSelectedToolNames = this.#runtimeSelectedToolNames; - this.#runtimeSelectedToolNames = new Set( - normalized.filter(name => !mounted.has(name) && !(name === "write" && transportWriteActive)), - ); - try { - await this.#applyActiveToolsByName(normalized); - } catch (error) { - this.#runtimeSelectedToolNames = previousRuntimeSelectedToolNames; - throw error; - } - } - - /** Cancel the local rollout-memory startup owned by this session. */ + /** Cancels the local rollout-memory startup owned by this session. */ cancelLocalMemoryStartup(): void { - this.#localMemoryStartupAbort?.abort(); - this.#localMemoryStartupAbort = undefined; + this.#memory.cancelLocalMemoryStartup(); } - /** Start a new local rollout-memory generation and cancel its predecessor. */ + /** Starts a new local rollout-memory generation and cancels its predecessor. */ beginLocalMemoryStartup(): AbortSignal { - this.cancelLocalMemoryStartup(); - const controller = new AbortController(); - this.#localMemoryStartupAbort = controller; - return controller.signal; + return this.#memory.beginLocalMemoryStartup(); } - /** Release the local startup slot if `signal` still owns it. */ + /** Releases the local startup slot if `signal` still owns it. */ endLocalMemoryStartup(signal: AbortSignal): void { - if (this.#localMemoryStartupAbort?.signal === signal) this.#localMemoryStartupAbort = undefined; + this.#memory.endLocalMemoryStartup(signal); } - async #disposeMemoryBackendState(consolidateMnemopi = true): Promise { - this.cancelLocalMemoryStartup(); - const hindsight = this.getHindsightSessionState(); - if (hindsight) { - try { - await hindsight.flushRetainQueue(); - } catch (error) { - logger.warn("Memory lifecycle: Hindsight flush failed", { error: String(error) }); - } - this.setHindsightSessionState(undefined); - hindsight.dispose(); - } - - const mnemopi = setMnemopiSessionState(this, undefined); - if (mnemopi) { - try { - await mnemopi.dispose({ consolidate: consolidateMnemopi }); - } catch (error) { - logger.warn("Memory lifecycle: Mnemopi dispose failed", { error: String(error) }); - } - } + /** Applies the selected memory backend to runtime state, tools, and prompt. */ + applyMemoryBackend(): Promise { + return this.#memory.applyMemoryBackend(); } - /** - * Apply the selected memory backend to runtime state, tools, and prompt. - * Concurrent settings changes run in order and settle before the next turn. - */ - async applyMemoryBackend(): Promise { - if (this.#isDisposed) return; - const transition = this.#memoryBackendTransition.then(() => this.#applyMemoryBackend()); - this.#memoryBackendTransition = transition.then( - () => undefined, - () => undefined, - ); - await transition; + /** Rebuilds the stable base prompt for the current tools and model. */ + refreshBaseSystemPrompt(): Promise { + return this.#tools.refreshBaseSystemPrompt(); } - async #applyMemoryBackend(): Promise { - if (this.#isDisposed) return; - try { - await this.#disposeMemoryBackendState(); - if (this.#memoryAgentDir && this.#memoryTaskDepth === 0 && !this.#isDisposed) { - const backend = await resolveMemoryBackend(this.settings); - await backend.start({ - session: this, - settings: this.settings, - modelRegistry: this.#modelRegistry, - agentDir: this.#memoryAgentDir, - taskDepth: this.#memoryTaskDepth, - }); - } - if (this.#isDisposed) return; - await this.#refreshMemoryTools(); - if (this.#isDisposed) return; - await this.refreshBaseSystemPrompt(); - } catch (error) { - await this.#disposeMemoryBackendState(false); - if (!this.#isDisposed) { - await this.#replaceMemoryTools([]).catch(refreshError => { - logger.warn("Failed to remove memory tools after backend apply error", { - error: String(refreshError), - }); - }); - } - throw error; - } + #buildSystemPromptForAgentStart(promptText: string): Promise { + return this.#tools.buildSystemPromptForAgentStart(promptText); } - async #refreshMemoryTools(): Promise { - const tools = (await this.#createMemoryTools?.()) ?? []; - await this.#replaceMemoryTools(tools); + /** Replaces connected MCP tools and enables them immediately. */ + refreshMCPTools(mcpTools: CustomTool[]): Promise { + return this.#tools.refreshMCPTools(mcpTools); } - async #replaceMemoryTools(tools: AgentTool[]): Promise { - const removed = new Set(MEMORY_BACKEND_TOOL_NAMES.filter(name => this.#builtInToolNames.has(name))); - const nextActive = this.getEnabledToolNames().filter(name => !removed.has(name)); - for (const name of removed) { - this.#toolRegistry.delete(name); - this.#builtInToolNames.delete(name); - } - - for (const tool of tools) { - if (!MEMORY_BACKEND_TOOL_NAMES.some(name => name === tool.name) || this.#toolRegistry.has(tool.name)) { - continue; - } - const wrapped = this.#wrapRuntimeTool(tool); - this.#toolRegistry.set(wrapped.name, wrapped); - this.#builtInToolNames.add(wrapped.name); - nextActive.push(wrapped.name); - } - await this.#applyActiveToolsByName([...new Set(nextActive)]); - } - - /** Rebuild the base system prompt using the current active tool set. */ - async refreshBaseSystemPrompt(): Promise { - if (this.#isDisposed || !this.#rebuildSystemPrompt) return; - const activeToolNames = this.getActiveToolNames(); - this.#setActiveToolNames?.(activeToolNames); - const previousBaseSystemPrompt = this.#baseSystemPrompt; - const built = await this.#rebuildSystemPrompt(activeToolNames, this.#toolRegistry); - if (this.#isDisposed) return; - this.#baseSystemPrompt = built.systemPrompt; - this.#baseSystemPromptBeforeMemoryPromotion = undefined; - if ( - previousBaseSystemPrompt.length !== this.#baseSystemPrompt.length || - previousBaseSystemPrompt.some((part, index) => part !== this.#baseSystemPrompt[index]) - ) { - this.#clearInheritedProviderPromptCacheKey(); - } - this.agent.setSystemPrompt(this.#baseSystemPrompt); - this.#promptModelKey = this.#currentPromptModelKey(); - // Refresh the cached signature so a subsequent `#applyActiveToolsByName` with - // the same tool set does not re-rebuild on top of the explicit refresh we - // just performed (and conversely, a different set forces a fresh rebuild). - const activeTools = activeToolNames - .map(name => this.#toolRegistry.get(name)) - .filter((tool): tool is AgentTool => tool != null); - this.#lastAppliedToolSignature = this.#computeAppliedToolSignature(activeToolNames, activeTools); - } - - async #buildSystemPromptForAgentStart(promptText: string): Promise { - const backend = await resolveMemoryBackend(this.settings); - if (!backend.beforeAgentStartPrompt) return this.#baseSystemPrompt; - - try { - const injected = await backend.beforeAgentStartPrompt(this, promptText); - if (!injected) return this.#baseSystemPrompt; - - const previousBaseSystemPrompt = this.#baseSystemPrompt; - try { - await this.refreshBaseSystemPrompt(); - } catch (refreshErr) { - logger.debug("Memory backend prompt refresh after beforeAgentStartPrompt failed", { - backend: backend.id, - error: String(refreshErr), - }); - } - - if ( - this.#baseSystemPrompt.length !== previousBaseSystemPrompt.length || - this.#baseSystemPrompt.some((part, index) => part !== previousBaseSystemPrompt[index]) - ) { - return this.#baseSystemPrompt; - } - - this.#baseSystemPromptBeforeMemoryPromotion ??= previousBaseSystemPrompt; - const stablePrompt = [...previousBaseSystemPrompt, injected]; - this.#baseSystemPrompt = stablePrompt; - this.agent.setSystemPrompt(stablePrompt); - return stablePrompt; - } catch (err) { - logger.debug("Memory backend beforeAgentStartPrompt failed", { - backend: backend.id, - error: String(err), - }); - return this.#baseSystemPrompt; - } - } - - /** - * Compose a stable signature for the inputs that `rebuildSystemPrompt` reads. - * Two calls producing identical signatures are guaranteed to produce identical - * system prompt bytes, so the rebuild can be skipped. - * - * The signature covers: - * 1. Active tool names in order (the prompt renders them in this order). - * 2. Active tool labels, descriptions, and wire-visible names — all are - * rendered into the prompt body (see `system-prompt.md` `{{label}}: \`{{name}}\`` - * and `toolPromptNames` in `buildSystemPrompt`). The wire name comes from - * `tool.customWireName` and overrides the internal name on the model wire - * (e.g. `edit` exposes itself as `apply_patch` to GPT-5 in apply_patch mode); - * a stale wire name would desync prompt guidance from actual tool routing. - * 3. When MCP discovery is on, every registry tool's name+label+description+ - * customWireName, since `rebuildSystemPrompt` summarizes discoverable MCP - * tools that are not in the active set. - * 4. MCP server instructions text (per server), since `rebuildSystemPrompt` - * embeds these in the appended prompt under "## MCP Server Instructions". - * A server upgrade can change instructions while keeping tools identical. - * - * Settings-driven tool metadata is covered automatically: built-in tools that - * depend on settings expose `description`/`label` via getters (see `TaskTool`, - * `SearchToolBm25Tool`, `EditTool`), and the signature reads them live on every - * call - so a settings flip that mutates the rendered string differs the signature - * the next time `#applyActiveToolsByName` runs. Do not refactor `describeTool` to - * cache per-tool strings without preserving this property. - * - * Inputs NOT covered: tool input schemas; memory instructions read from disk; - * and SDK-init-time closure constants in `sdk.ts` (`inlineToolDescriptors`, - * `eagerTasks`, `intentField`, `mcpDiscoveryEnabled`, `secretsEnabled`). The - * closure-captured ones cannot change at runtime regardless of skip behavior. - * For everything else, callers must explicitly call `refreshBaseSystemPrompt()` - * after side-effecting changes; see e.g. the memory hooks and - * `#syncAfterModelChange`. - * - * The current calendar date IS covered (appended as a segment) because - * `buildSystemPrompt` injects it into the prompt body (`Today is '{{date}}'`). - * Without this, a session spanning midnight with only tool-stable MCP - * reconnects would keep yesterday's date indefinitely. - */ - #computeAppliedToolSignature(toolNames: string[], tools: AgentTool[]): string { - // Order-preserving join: any reorder must produce a different signature so - // the rebuild fires and the new tool list reaches the API. - const nameSegment = toolNames.join("\u0001"); - const describeTool = (tool: AgentTool): string => - `${tool.name}=${tool.label ?? ""}|${tool.description ?? ""}|${tool.customWireName ?? ""}`; - const descriptionSegment = tools.map(describeTool).join("\u0002"); - let instructionsSegment = ""; - const serverInstructions = this.#getMcpServerInstructions?.(); - if (serverInstructions && serverInstructions.size > 0) { - // Sort by server name so transport flap order does not perturb the signature. - const entries: string[] = []; - for (const [server, instructions] of serverInstructions) { - entries.push(`${server}=${instructions}`); - } - entries.sort(); - instructionsSegment = entries.join("\u0006"); - } - // The xd:// device inventory is deliberately NOT part of the signature: - // a mount/unmount announces itself via `#notifyXdevMountDelta` instead of - // rewriting the system prompt, so MCP connects/disconnects keep the - // prompt (and its provider cache prefix) byte-stable. Rebuilds triggered - // by other inputs pick up the current device docs opportunistically. - const date = this.#getLocalCalendarDate(); - return `${nameSegment}\u0003${descriptionSegment}\u0007${instructionsSegment}|${date}`; - } - - /** - * Replace MCP tools in the registry and enable them immediately. Every - * connected MCP tool becomes available (mounted under `xd://` when that - * transport is active, else top-level). Lets `/mcp add/remove/reauth` take - * effect without restarting the session. - */ - async refreshMCPTools(mcpTools: CustomTool[]): Promise { - const existingNames = Array.from(this.#toolRegistry.keys()); - const previousMcpTools = new Map( - existingNames.flatMap(name => { - const tool = this.#toolRegistry.get(name); - return isMCPToolName(name) && tool ? [[name, tool] as const] : []; - }), - ); - for (const name of existingNames) { - if (isMCPToolName(name)) { - this.#toolRegistry.delete(name); - } - } - - const getCustomToolContext = (): CustomToolContext => ({ - sessionManager: this.sessionManager, - modelRegistry: this.#modelRegistry, - model: this.model, - isIdle: () => !this.isStreaming, - hasQueuedMessages: () => this.queuedMessageCount > 0, - abort: () => { - this.agent.abort(); - }, - settings: this.settings, - localProtocolOptions: this.#localProtocolOptions(), - }); - - for (const customTool of mcpTools) { - const wrapped = wrapToolWithMetaNotice(CustomToolAdapter.wrap(customTool, getCustomToolContext) as AgentTool); - const finalTool = ( - this.#extensionRunner ? new ExtensionToolWrapper(wrapped, this.#extensionRunner) : wrapped - ) as AgentTool; - this.#toolRegistry.set(finalTool.name, finalTool); - } - - // Every connected MCP tool is selected; centralized repartitioning owns - // presentation pins and write-transport activation/removal. - const nextActive = [...new Set([...this.#getActiveNonMCPToolNames(), ...mcpTools.map(tool => tool.name)])]; - try { - await this.#applyActiveToolsByName(nextActive); - } catch (error) { - for (const name of this.#toolRegistry.keys()) { - if (isMCPToolName(name)) this.#toolRegistry.delete(name); - } - for (const [name, tool] of previousMcpTools) this.#toolRegistry.set(name, tool); - throw error; - } - } - - /** - * Replace RPC host-owned tools and refresh the active tool set before the next model call. - */ - async refreshRpcHostTools(rpcTools: AgentTool[]): Promise { - const nextToolNames = rpcTools.map(tool => tool.name); - const uniqueToolNames = new Set(nextToolNames); - if (uniqueToolNames.size !== nextToolNames.length) { - throw new Error("RPC host tool names must be unique"); - } - - for (const name of uniqueToolNames) { - if (this.#toolRegistry.has(name) && !this.#rpcHostToolNames.has(name)) { - throw new Error(`RPC host tool "${name}" conflicts with an existing tool`); - } - } - - const previousRpcHostToolNames = new Set(this.#rpcHostToolNames); - const previousActiveToolNames = this.getEnabledToolNames(); - const previousRpcHostTools = new Map( - [...previousRpcHostToolNames].flatMap(name => { - const tool = this.#toolRegistry.get(name); - return tool ? [[name, tool] as const] : []; - }), - ); - for (const name of previousRpcHostToolNames) { - this.#toolRegistry.delete(name); - } - this.#rpcHostToolNames.clear(); - - for (const tool of rpcTools) { - const metaWrapped = wrapToolWithMetaNotice(tool); - const finalTool = ( - this.#extensionRunner ? new ExtensionToolWrapper(metaWrapped, this.#extensionRunner) : metaWrapped - ) as AgentTool; - this.#toolRegistry.set(finalTool.name, finalTool); - this.#rpcHostToolNames.add(finalTool.name); - } - - const activeNonRpcToolNames = previousActiveToolNames.filter(name => !previousRpcHostToolNames.has(name)); - const preservedRpcToolNames = previousActiveToolNames.filter( - name => previousRpcHostToolNames.has(name) && this.#rpcHostToolNames.has(name), - ); - const autoActivatedRpcToolNames = rpcTools - .filter(tool => !tool.hidden && !previousRpcHostToolNames.has(tool.name)) - .map(tool => tool.name); - try { - await this.#applyActiveToolsByName( - Array.from(new Set([...activeNonRpcToolNames, ...preservedRpcToolNames, ...autoActivatedRpcToolNames])), - ); - } catch (error) { - for (const name of this.#rpcHostToolNames) this.#toolRegistry.delete(name); - this.#rpcHostToolNames = previousRpcHostToolNames; - for (const [name, tool] of previousRpcHostTools) this.#toolRegistry.set(name, tool); - throw error; - } + /** Replaces host-owned RPC tools before the next model call. */ + refreshRpcHostTools(rpcTools: AgentTool[]): Promise { + return this.#tools.refreshRpcHostTools(rpcTools); } /** Whether auto-compaction is currently running */ get isCompacting(): boolean { - return this.#autoCompactionAbortController !== undefined || this.#compactionAbortController !== undefined; + return this.#maintenance.isCompacting; + } + + /** Strip image content from the current branch and persist the rewrite. */ + dropImages(): Promise<{ removed: number }> { + return this.#maintenance.dropImages(); + } + + /** Reduce stored context with the selected shake strategy. */ + shake(mode: ShakeMode, opts: { config?: ShakeConfig; signal?: AbortSignal } = {}): Promise { + return this.#maintenance.shake(mode, opts); + } + + /** Compact the active session history. */ + compact(customInstructions?: string, options?: CompactOptions): Promise { + return this.#maintenance.compact(customInstructions, options); + } + + /** Cancel active manual, automatic, and handoff maintenance. */ + abortCompaction(): void { + this.#maintenance.abortCompaction(); + } + + /** Trigger idle compaction through the automatic maintenance flow. */ + async runIdleCompaction(): Promise { + await this.#maintenance.runIdleCompaction(); + } + + /** Toggle automatic compaction. */ + setAutoCompactionEnabled(enabled: boolean): void { + this.#maintenance.setAutoCompactionEnabled(enabled); + } + + /** Whether automatic compaction is enabled. */ + get autoCompactionEnabled(): boolean { + return this.#maintenance.autoCompactionEnabled; } /** @@ -8246,24 +3855,11 @@ export class AgentSession { /** Latest image attachments addressable by tools as `Image #N` or `attachment://N`. */ getImageAttachments(): { label: string; uri: string; image: ImageContent }[] { - for (let i = this.agent.state.messages.length - 1; i >= 0; i--) { - const message = this.agent.state.messages[i]; - if (!message || (message.role !== "user" && message.role !== "developer") || !Array.isArray(message.content)) { - continue; - } - const images = message.content.filter((part): part is ImageContent => part.type === "image"); - if (images.length === 0) continue; - return images.map((image, index) => ({ - label: `Image #${index + 1}`, - uri: `attachment://${index + 1}`, - image, - })); - } - return []; + return this.#providerBoundary.getImageAttachments(); } buildDisplaySessionContext(): SessionContext { - return deobfuscateSessionContext(this.sessionManager.buildSessionContext(), this.#obfuscator); + return this.#providerBoundary.buildDisplaySessionContext(); } /** @@ -8278,164 +3874,37 @@ export class AgentSession { buildTranscriptSessionContext( options?: Pick, ): SessionContext { - return deobfuscateSessionContext( - this.sessionManager.buildSessionContext({ - transcript: true, - collapseCompactedHistory: options?.collapseCompactedHistory, - keepDanglingToolCalls: options?.keepDanglingToolCalls, - }), - this.#obfuscator, - true, - ); + return this.#providerBoundary.buildTranscriptSessionContext(options); } #obfuscateTextForProvider(text: string | undefined): string | undefined { - if (!text || !this.#obfuscator?.hasSecrets()) return text; - return this.#obfuscator.obfuscate(text); + return this.#providerBoundary.obfuscateText(text); } #obfuscatePreparationForProvider(preparation: CompactionPreparation): CompactionPreparation { - if (!this.#obfuscator?.hasSecrets()) return preparation; - const previousSummary = this.#obfuscateTextForProvider(preparation.previousSummary); - // `compact()` folds the prior snapcompact archive's plaintext into the - // summarization prompt on the snapcompact→context-full transition, so the - // archive's text regions must be redacted alongside the summary. Only the - // `snapcompact` slot's text is rewritten; every other preserveData key — - // notably the OpenAI remote-compaction `encrypted_content` replay state — is - // opaque provider-replay data and stays byte-identical. - const previousPreserveData = this.#obfuscatePreservedArchiveText(preparation.previousPreserveData); - if ( - previousSummary === preparation.previousSummary && - previousPreserveData === preparation.previousPreserveData - ) { - return preparation; - } - return { ...preparation, previousSummary, previousPreserveData }; - } - - /** Redact secrets in the persisted snapcompact archive's plaintext regions - * ({@link snapcompact.archiveSourceText}'s `text`/`textHead`/`textTail`) so the - * snapcompact→context-full migration in `compact()` cannot ship raw archived - * user/tool text to the provider. Frames and every non-`snapcompact` key pass - * through byte-identical; the same reference is returned when nothing changes. */ - #obfuscatePreservedArchiveText( - preserveData: Record | undefined, - ): Record | undefined { - const obfuscator = this.#obfuscator; - if (!obfuscator?.hasSecrets() || !preserveData || !snapcompact.getPreservedArchive(preserveData)) { - return preserveData; - } - const slot = preserveData[snapcompact.PRESERVE_KEY] as Record; - const obfuscated: Record = { ...slot }; - let changed = false; - for (const key of ["text", "textHead", "textTail"] as const) { - const value = slot[key]; - if (typeof value !== "string" || value.length === 0) continue; - const next = obfuscator.obfuscate(value); - if (next === value) continue; - obfuscated[key] = next; - changed = true; - } - return changed ? { ...preserveData, [snapcompact.PRESERVE_KEY]: obfuscated } : preserveData; + return this.#providerBoundary.obfuscateCompactionPreparation(preparation); } #deobfuscateFromProvider(text: string): string { - if (!this.#obfuscator?.hasSecrets()) return text; - return this.#obfuscator.deobfuscate(text); + return this.#providerBoundary.deobfuscateText(text); } #deobfuscatedProviderTextReadyForDelta(text: string): string { - const deobfuscated = this.#deobfuscateFromProvider(text); - if (!this.#obfuscator?.hasSecrets()) return deobfuscated; - return stripPendingSecretPlaceholderSuffix(deobfuscated); + return this.#providerBoundary.deobfuscateDelta(text); } #convertToLlmForSideRequest(messages: AgentMessage[]): Message[] { - const converted = convertToLlm(messages); - return this.#obfuscator?.hasSecrets() ? obfuscateMessages(this.#obfuscator, converted) : converted; + return this.#providerBoundary.convertToLlmForSideRequest(messages); } /** Convert session messages using the same pre-LLM pipeline as the active session. */ async convertMessagesToLlm(messages: AgentMessage[], signal?: AbortSignal): Promise { - const transformedMessages = await this.#transformContext(messages, signal); - return await this.#convertToLlm(transformedMessages); + return await this.#providerBoundary.convertMessagesToLlm(messages, signal); } /** Apply session-level stream hooks to a direct side request. */ prepareSimpleStreamOptions(options: SimpleStreamOptions, provider = "anthropic"): SimpleStreamOptions { - const sessionOnPayload = this.#onPayload; - const sessionOnResponse = this.#onResponse; - const sessionMetadata = this.agent.metadataForProvider(provider); - const sessionOnSseEvent = this.#onSseEvent; - const openrouterRoutingPreset = - provider === "openrouter" ? this.settings.get("providers.openrouterVariant") : "default"; - const openrouterVariant = - openrouterRoutingPreset !== "default" && options.openrouterVariant === undefined - ? openrouterRoutingPreset - : undefined; - const antigravityEndpointMode = - provider === "google-antigravity" ? this.settings.get("providers.antigravityEndpoint") : undefined; - - const preparedOptions: SimpleStreamOptions = { - ...options, - ...(openrouterVariant !== undefined && { openrouterVariant }), - ...(antigravityEndpointMode !== undefined && { antigravityEndpointMode }), - maxInFlightRequests: validateProviderMaxInFlightRequests( - options.maxInFlightRequests ?? this.settings.get("providers.maxInFlightRequests"), - ), - loopGuard: { - enabled: this.settings.get("model.loopGuard.enabled"), - checkAssistantContent: this.settings.get("model.loopGuard.checkAssistantContent"), - ...options.loopGuard, - }, - }; - - // Stamp session metadata (e.g. user_id={session_id}) onto direct-call requests so - // they share the same session bucket as Agent.prompt-routed requests on Anthropic - // OAuth. Caller-provided metadata wins so explicit overrides are respected. - if (sessionMetadata && !options.metadata) { - preparedOptions.metadata = sessionMetadata; - } - - if (sessionOnPayload) { - if (!options.onPayload) { - preparedOptions.onPayload = sessionOnPayload; - } else { - const requestOnPayload = options.onPayload; - preparedOptions.onPayload = async (payload, model) => { - const sessionPayload = await sessionOnPayload(payload, model); - const sessionResolvedPayload = sessionPayload ?? payload; - const requestPayload = await requestOnPayload(sessionResolvedPayload, model); - return requestPayload ?? sessionResolvedPayload; - }; - } - } - - if (sessionOnResponse) { - if (!options.onResponse) { - preparedOptions.onResponse = sessionOnResponse; - } else { - const requestOnResponse = options.onResponse; - preparedOptions.onResponse = async (response, model) => { - await sessionOnResponse(response, model); - await requestOnResponse(response, model); - }; - } - } - - if (sessionOnSseEvent) { - if (!options.onSseEvent) { - preparedOptions.onSseEvent = sessionOnSseEvent; - } else { - const requestOnSseEvent = options.onSseEvent; - preparedOptions.onSseEvent = (event, model) => { - sessionOnSseEvent(event, model); - requestOnSseEvent(event, model); - }; - } - } - - return preparedOptions; + return this.#providerBoundary.prepareSimpleStreamOptions(options, provider); } /** Current steering mode */ @@ -8463,11 +3932,7 @@ export class AgentSession { return this.#activeProviderSessionId(); } getEvalSessionId(): string | null { - if (this.#parentEvalSessionId !== undefined) return this.#parentEvalSessionId; - return defaultEvalSessionId({ - cwd: this.sessionManager.getCwd(), - getSessionFile: () => this.sessionManager.getSessionFile() ?? null, - }); + return this.#eval.getSessionId(); } /** Current session display name, if set */ @@ -8477,7 +3942,7 @@ export class AgentSession { /** Scoped models for cycling (from --models flag) */ get scopedModels(): ReadonlyArray<{ model: Model; thinkingLevel?: ThinkingLevel }> { - return this.#scopedModels; + return this.#models.scopedModels; } /** Prompt templates */ @@ -8487,7 +3952,7 @@ export class AgentSession { /** Prewalk state, if armed and active */ getPrewalkState(): Prewalk | undefined { - return this.#prewalk; + return this.#prewalk.state; } setPlanModeState(state: PlanModeState | undefined): void { @@ -8548,18 +4013,7 @@ export class AgentSession { setClientBridge(bridge: ClientBridge | undefined): void { this.#clientBridge = bridge; - this.#acpPermissionDecisions.clear(); - const activeToolNames = this.getActiveToolNames(); - const activeTools = activeToolNames - .map(name => this.#toolRegistry.get(name)) - .filter((tool): tool is AgentTool => tool !== undefined) - .map(tool => this.#wrapToolForAcpPermission(tool)); - this.agent.setTools(activeTools); - const mountedTools = [...this.#mountedXdevToolNames] - .map(name => this.#toolRegistry.get(name)) - .filter((tool): tool is AgentTool => tool !== undefined) - .map(tool => this.#wrapToolForAcpPermission(tool)); - this.#xdevRegistry?.reconcile(mountedTools); + this.#tools.refreshAcpPermissionGates(); } #clearCheckpointRuntimeState(): void { @@ -8679,7 +4133,7 @@ export class AgentSession { } resolveRoleModel(role: string): Model | undefined { - return this.#resolveRoleModelFull(role, this.#modelRegistry.getAvailable(), this.model).model; + return this.#models.resolveRoleModel(role); } /** @@ -8688,7 +4142,7 @@ export class AgentSession { * from role configuration (e.g., "anthropic/claude-sonnet-4-5:xhigh"). */ resolveRoleModelWithThinking(role: string): ResolvedModelRoleValue { - return this.#resolveRoleModelFull(role, this.#modelRegistry.getAvailable(), this.model); + return this.#models.resolveRoleModelWithThinking(role); } /** @@ -8696,23 +4150,7 @@ export class AgentSession { * picker selects a model already assigned to a configured role. */ resolveTemporaryModelThinkingLevel(model: Model): ConfiguredThinkingLevel | undefined { - const availableModels = this.#modelRegistry.getAvailable(); - if (availableModels.length === 0) return undefined; - - const matchPreferences = getModelMatchPreferences(this.settings); - for (const role of getKnownRoleIds(this.settings)) { - const roleValue = this.settings.getModelRole(role); - if (!roleValue) continue; - - const resolved = resolveModelRoleValue(roleValue, availableModels, { - settings: this.settings, - matchPreferences, - }); - if (!resolved.explicitThinkingLevel || resolved.thinkingLevel === undefined || !resolved.model) continue; - if (modelsAreEqual(resolved.model, model)) return resolved.thinkingLevel; - } - - return undefined; + return this.#models.resolveTemporaryModelThinkingLevel(model); } get promptTemplates(): ReadonlyArray { @@ -8887,81 +4325,15 @@ export class AgentSession { return normalizeModelContextImages(images, { model: this.model }); } - /** - * Build a hidden companion message describing image attachments for a text-only - * model. Each image is saved under local:// and a vision-capable model describes - * it; the descriptions are returned as a `display: false` custom message (so the - * model reads them but the TUI does not render the blob) carrying one - * `…` block per image. Returns `undefined` when - * the active model already accepts images, the feature is disabled, or no - * description could be produced. Never throws. - */ - async #buildImageDescriptionNotice( + #buildImageDescriptionNotice( normalizedImages: ImageContent[], signal?: AbortSignal, ): Promise { - const model = this.model; - const shouldDescribe = - !!model && - !model.input.includes("image") && - !this.settings.get("images.blockImages") && - this.settings.get("images.describeForTextModels"); - if (!shouldDescribe || !model) { - return undefined; - } - let blocks: TextContent[]; - try { - blocks = await describeAttachedImagesForTextModel( - normalizedImages, - { - activeModel: model, - modelRegistry: this.#modelRegistry, - settings: this.settings, - localProtocolOptions: this.#localProtocolOptions(), - activeModelString: formatModelString(model), - telemetryConfig: this.agent.telemetry, - sessionId: this.sessionId, - }, - signal, - ); - } catch (err) { - logger.warn("image attachment vision fallback failed; image left undescribed", { - error: err instanceof Error ? err.message : String(err), - }); - return undefined; - } - if (blocks.length === 0) { - return undefined; - } - return { - role: "custom", - customType: IMAGE_ATTACHMENT_DESCRIPTION_TYPE, - content: blocks, - display: false, - attribution: "user", - timestamp: Date.now(), - }; + return this.#providerBoundary.buildImageDescriptionNotice(normalizedImages, signal); } - async #normalizeMessageContentImages( - content: string | (TextContent | ImageContent)[], - ): Promise { - if (typeof content === "string") return content; - const images = content.filter((part): part is ImageContent => part.type === "image"); - if (images.length === 0) return content; - const normalizedImages = await this.#normalizeImagesForModel(images); - if (!normalizedImages) return content; - let imageIndex = 0; - return content.map(part => (part.type === "image" ? normalizedImages[imageIndex++]! : part)); - } - - async #normalizeAgentMessageImages(message: T): Promise { - if (!("content" in message)) return message; - const content = message.content; - if (typeof content !== "string" && !Array.isArray(content)) return message; - const normalized = await this.#normalizeMessageContentImages(content as string | (TextContent | ImageContent)[]); - if (normalized === content) return message; - return { ...message, content: normalized } as T; + #normalizeAgentMessageImages(message: T): Promise { + return this.#providerBoundary.normalizeAgentMessageImages(message); } #magicKeywordEnabled(keyword: "orchestrate" | "ultrathink" | "workflow"): boolean { @@ -9063,7 +4435,7 @@ export class AgentSession { // re-enables advisor auto-resume that a prior user interrupt suppressed. // Agent-initiated synthetic prompts (auto-continue, plan, reminders) do not. if (options?.userInitiated ?? !options?.synthetic) { - this.#advisorAutoResumeSuppressed = false; + this.#advisors.autoResumeSuppressed = false; this.#planModeReminderCount = 0; this.#planModeReminderAwaitingProgress = false; // A user turn owns the next decision; drop a queued forced choice from @@ -9092,9 +4464,9 @@ export class AgentSession { // Skip eager preludes when the user has already queued a directive const hasPendingUserDirective = this.#toolChoiceQueue.inspect().includes("user-force"); const eagerTodoPrelude = - !options?.synthetic && !hasPendingUserDirective ? this.#createEagerTodoPrelude(expandedText) : undefined; + !options?.synthetic && !hasPendingUserDirective ? this.#todo.createEagerTodoPrelude(expandedText) : undefined; const eagerTaskPrelude = - !options?.synthetic && !hasPendingUserDirective ? this.#createEagerTaskPrelude(expandedText) : undefined; + !options?.synthetic && !hasPendingUserDirective ? this.#todo.createEagerTaskPrelude(expandedText) : undefined; const normalizedImages = await this.#normalizeImagesForModel(options?.images); const userContent: (TextContent | ImageContent)[] = [{ type: "text", text: expandedText }]; @@ -9220,19 +4592,15 @@ export class AgentSession { const generation = this.#promptGeneration; try { // Flush any pending bash messages before the new prompt - await this.#flushPendingBashMessages(); - this.#flushPendingPythonMessages(); - this.#flushPendingIrcAsides(); + await this.#bash.flushPending(); + this.#eval.flushPending(); + this.#irc.flushPending(); - // Reset todo reminder count on new user prompt - this.#todoReminderCount = 0; - this.#todoReminderAwaitingProgress = false; - this.#mutationsSinceLastTodoTouch = 0; - this.#midRunNudgeCount = 0; + this.#todo.resetCycle(); this.#resetPromptMaintenanceState(); - this.#acceptTerminalEmptyStopForPrompt = options?.acceptTerminalEmptyStop === true; + this.#recovery.setAcceptTerminalEmptyStop(options?.acceptTerminalEmptyStop === true); - await this.#maybeRestoreRetryFallbackPrimary(); + await this.#recovery.maybeRestoreRetryFallbackPrimary(); // Validate model if (!this.model) { @@ -9262,10 +4630,10 @@ export class AgentSession { !options?.skipCompactionCheck && (lastAssistant.stopReason === "error" || lastAssistant.stopReason === "length") ) { - await this.#checkCompaction(lastAssistant, false, false, false); + await this.#maintenance.checkCompaction(lastAssistant, false, false, false); } - await this.#armPlanYoloIfNeeded(); + await this.#prewalk.armPlanYoloIfNeeded(); // Build messages array (session context, eager todo prelude, then active prompt message) const messages: AgentMessage[] = []; @@ -9327,7 +4695,7 @@ export class AgentSession { // prompt when disposal began during the backend-transition await, where // resuming would start a turn on a torn-down session. const disposingBeforeTransition = this.#isDisposed; - await this.#memoryBackendTransition; + await this.#memory.transition; if ((this.#isDisposed && !disposingBeforeTransition) || this.#promptGeneration !== generation) return; const beforeAgentStartSystemPrompt = await this.#buildSystemPromptForAgentStart(expandedText); @@ -9382,14 +4750,14 @@ export class AgentSession { // before the model request. Synthetic/tool-continuation turns (developer/ // custom roles) and non-auto sessions are skipped. Never blocks the turn — // failures fall back to a concrete level inside the helper. - if (this.#autoThinking && message.role === "user") { - await this.#applyAutoThinkingLevel(expandedText, generation); + if (this.isAutoThinking && message.role === "user") { + await this.#models.applyAutoThinkingLevel(expandedText, generation); if (this.#promptGeneration !== generation) { return; } } - await this.#runPrePromptCompactionIfNeeded(messages); + await this.#maintenance.runPrePromptCompactionIfNeeded(messages); if (this.#promptGeneration !== generation) { return; } @@ -9403,15 +4771,15 @@ export class AgentSession { nonMessageTokens + this.messages.reduce((sum, msg) => sum + estimateTokens(msg), 0) + messages.reduce((sum, msg) => sum + estimateTokens(msg), 0); - this.#setPendingContextSnapshot({ + this.#stats.setPendingSnapshot({ promptTokens, nonMessageTokens, cutoffCount: this.messages.length + messages.length, }); try { - await this.#promptAgentWithIdleRetry(messages, agentPromptOptions); + await this.#recovery.promptAgentWithIdleRetry(messages, agentPromptOptions); } finally { - this.#setPendingContextSnapshot(undefined); + this.#stats.setPendingSnapshot(undefined); } if (!options?.skipPostPromptRecoveryWait) { await this.#waitForPostPromptRecovery(generation); @@ -9632,7 +5000,7 @@ export class AgentSession { // A queued user message (RPC/SDK/collab steer or follow-up, or a typed message // while streaming) is a deliberate resume; re-enable advisor auto-resume that // a user interrupt suppressed. - this.#advisorAutoResumeSuppressed = false; + this.#advisors.autoResumeSuppressed = false; const normalizedImages = await this.#normalizeImagesForModel(images); const content: (TextContent | ImageContent)[] = [{ type: "text", text }]; if (normalizedImages?.length) { @@ -9704,7 +5072,7 @@ export class AgentSession { // (#advisorAutoResumeSuppressed, cleared on the next user prompt): the user stopped, so their // queued follow-up waits for an explicit resume — even if an interleaving IRC wake turn has // since left a provider-valid tail. - if (this.#advisorAutoResumeSuppressed) return false; + if (this.#advisors.autoResumeSuppressed) return false; // Follow-up-only resume has no steer to inject, so Agent.continue() continues from the // existing context tail — which must itself be a valid provider tail. An injected // non-conversational tail (advisor card → `developer`, bash/python execution) would make @@ -9813,11 +5181,11 @@ export class AgentSession { if (acceptTerminalEmptyStop) { this.#resetPromptMaintenanceState(); } - this.#acceptTerminalEmptyStopForPrompt = acceptTerminalEmptyStop; + this.#recovery.setAcceptTerminalEmptyStop(acceptTerminalEmptyStop); await this.agent.prompt(message); await this.#waitForPostPromptRecovery(); } finally { - this.#acceptTerminalEmptyStopForPrompt = false; + this.#recovery.setAcceptTerminalEmptyStop(false); this.#endInFlight(); } } @@ -10086,54 +5454,29 @@ export class AgentSession { } get skillsSettings(): SkillsSettings | undefined { - return this.#skillsSettings; + return this.#tools.skillsSettings; } /** Skills loaded by SDK (empty if --no-skills or skills: [] was passed) */ get skills(): readonly Skill[] { - return this.#skills; + return this.#tools.skills; } /** Skill loading warnings captured by SDK */ get skillWarnings(): readonly SkillWarning[] { - return this.#skillWarnings; + return this.#tools.skillWarnings; } getTodoPhases(): TodoPhase[] { - return this.#cloneTodoPhases(this.#todoPhases); + return this.#todo.phases; } setTodoPhases(phases: TodoPhase[]): void { - this.#todoPhases = this.#cloneTodoPhases(phases); - } - - #isTodoInitResult(details: Record, toolCallId: string | undefined): boolean { - const detailOp = getStringProperty(details, "op"); - if (detailOp) return detailOp === "init"; - if (!toolCallId) return false; - for (let i = this.agent.state.messages.length - 1; i >= 0; i--) { - const message = this.agent.state.messages[i]; - if (!message) continue; - const op = toolCallOpFromMessage(message, toolCallId); - if (op) return op === "init"; - } - return false; + this.#todo.setPhases(phases); } #buildReplanTitleContext(): string { - const turns: TitleConversationTurn[] = []; - for ( - let i = this.agent.state.messages.length - 1; - i >= 0 && turns.length < REPLAN_TITLE_CONTEXT_TURN_LIMIT; - i-- - ) { - const message = this.agent.state.messages[i]; - if (!message) continue; - const turn = titleConversationTurnFromMessage(message); - if (turn) turns.push(turn); - } - turns.reverse(); - return formatTitleConversationContext(turns); + return buildReplanTitleContext(this.agent.state.messages); } #scheduleReplanTitleRefresh(): void { @@ -10208,28 +5551,6 @@ export class AgentSession { this.#titleSystemPrompt = prompt; } - #syncTodoPhasesFromBranch(): void { - const phases = getLatestTodoPhasesFromEntries(this.sessionManager.getBranch()); - this.setTodoPhases(phases); - } - - #cloneTodoPhases(phases: TodoPhase[]): TodoPhase[] { - return phases.map(phase => ({ - name: phase.name, - tasks: phase.tasks.map(task => - task.blocker !== undefined - ? { content: task.content, status: task.status, blocker: task.blocker } - : { content: task.content, status: task.status }, - ), - })); - } - - // Auto-clear of completed/abandoned tasks was removed: the timer-driven - // splice mutated canonical `#todoPhases` between tool calls, so the model - // observed phase totals shrinking ("5 → 4") after marking tasks done. The - // `tasks.todoClearDelay` setting is now inert; completed tasks survive - // until the next explicit `todo` call removes them via `rm`/`drop`. - /** * Abort current operation and wait for agent to become idle. * @@ -10246,7 +5567,7 @@ export class AgentSession { }): Promise { const userInterrupt = options?.reason === USER_INTERRUPT_LABEL; this.#pendingAbortErrorId = userInterrupt ? AIError.create(AIError.Flag.UserInterrupt) : undefined; - if (userInterrupt) this.#advisorAutoResumeSuppressed = true; + if (userInterrupt) this.#advisors.autoResumeSuppressed = true; // Pull advisor concerns out of the steer/follow-up queues before any await so // the post-abort stranded-message drain can't auto-resume the run on them. // They are re-recorded as visible advice once the agent settles (below). @@ -10267,7 +5588,7 @@ export class AgentSession { // auto-compaction MUST still be cancelled, though: otherwise a // background maintenance pass races the manual run and both // appendCompaction/replaceMessages, double-rewriting session history. - this.#autoCompactionAbortController?.abort(); + this.#maintenance.abortAutomaticCompaction(); } else { this.abortCompaction(); } @@ -10339,8 +5660,8 @@ export class AgentSession { await this.abort(); this.#cancelOwnAsyncJobs(); this.#closeAllProviderSessions("new session"); - await this.#flushPendingBashMessages(); - const bashTransition = this.#beginBashSessionTransition({ persistDetached: options?.drop !== true }); + await this.#bash.flushPending(); + const bashTransition = this.#bash.beginSessionTransition({ persistDetached: options?.drop !== true }); let sessionTransitioned = false; try { this.agent.reset(); @@ -10350,11 +5671,7 @@ export class AgentSession { // running advisor turn could otherwise finish, emit `message_end`, and recreate // `/__advisor.jsonl`. #resetAdvisorSessionState (after newSession) re-primes // the advisor and re-attaches the feed at the new session's path. - for (const a of this.#advisors) { - a.agentUnsubscribe?.(); - a.agentUnsubscribe = undefined; - await a.recorder.close(); - } + await this.#advisors.detachAndCloseRecorders(); try { await this.sessionManager.dropSession(previousSessionFile); } catch (err) { @@ -10367,10 +5684,10 @@ export class AgentSession { ...options, additionalDirectories: this.settings.get("workspace.additionalDirectories"), }); - this.#markBashSessionTransition(bashTransition); + this.#bash.markSessionTransition(bashTransition); sessionTransitioned = true; } finally { - this.#finishBashSessionTransition(bashTransition, sessionTransitioned); + this.#bash.finishSessionTransition(bashTransition, sessionTransitioned); } this.#clearCheckpointRuntimeState(); @@ -10378,22 +5695,18 @@ export class AgentSession { this.#freshProviderSessionId = undefined; this.#clearInheritedProviderPromptCacheKey(); this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); - await this.#resetMemoryContextForNewTranscript(); + this.#memory.rekeyForCurrentSessionId(); + await this.#memory.resetContextForNewTranscript(); this.#pendingNextTurnMessages = []; this.#scheduledHiddenNextTurnGeneration = undefined; this.sessionManager.appendThinkingLevelChange(this.thinkingLevel, this.configuredThinkingLevel()); - this.sessionManager.appendServiceTierChange(this.#serviceTierEntry()); + this.sessionManager.appendServiceTierChange(this.#models.serviceTierEntry()); - this.#todoReminderCount = 0; - this.#todoReminderAwaitingProgress = false; - this.#mutationsSinceLastTodoTouch = 0; - this.#midRunNudgeCount = 0; + this.#todo.resetCycle(); this.#planReferenceSent = false; this.#planReferencePath = "local://PLAN.md"; - this.#resetAdvisorSessionState(); + this.#advisors.resetSessionState(); this.#reconnectToAgent(); // The workspace-roots block must reflect the new session's directory set, // not the previous session's — refresh before the next turn goes out. @@ -10441,25 +5754,25 @@ export class AgentSession { } } - await this.#flushPendingBashMessages(); + await this.#bash.flushPending(); // Flush current session to ensure all entries are written await this.sessionManager.flush(); - const bashTransition = this.#beginBashSessionTransition(); + const bashTransition = this.#bash.beginSessionTransition(); // Fork the session (creates new session file with same entries) let forkResult: { oldSessionFile: string; newSessionFile: string } | undefined; try { forkResult = await this.sessionManager.fork(); } catch (error) { - this.#finishBashSessionTransition(bashTransition, false); + this.#bash.finishSessionTransition(bashTransition, false); throw error; } if (!forkResult) { - this.#finishBashSessionTransition(bashTransition, false); + this.#bash.finishSessionTransition(bashTransition, false); return false; } - this.#markBashSessionTransition(bashTransition); - this.#finishBashSessionTransition(bashTransition, true); + this.#bash.markSessionTransition(bashTransition); + this.#bash.finishSessionTransition(bashTransition, true); // Copy artifacts directory if it exists const oldArtifactDir = forkResult.oldSessionFile.slice(0, -6); @@ -10484,9 +5797,8 @@ export class AgentSession { this.#freshProviderSessionId = undefined; this.#adoptInheritedProviderPromptCacheKey(); this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); - await this.#resetMemoryContextForNewTranscript(); + this.#memory.rekeyForCurrentSessionId(); + await this.#memory.resetContextForNewTranscript(); // Emit session_switch event with reason "fork" to hooks if (this.#extensionRunner) { @@ -10529,519 +5841,84 @@ export class AgentSession { currentContextTokens?: number; }, ): Promise<{ switched: boolean }> { - const previousEditMode = this.#resolveActiveEditMode(); - if (!this.#modelRegistry.hasConfiguredAuth(model)) { - throw new Error(`No API key for ${model.provider}/${model.id}`); - } - - const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model); - - this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); - this.#clearActiveRetryFallback(); - this.#setModelWithProviderSessionReset(targetModel); - this.sessionManager.appendModelChange(`${targetModel.provider}/${targetModel.id}`, role); - if (options?.persist) { - this.settings.setModelRole( - role, - this.#formatRoleModelValue(role, targetModel, options.selector, options.thinkingLevel), - ); - } - this.settings.getStorage()?.recordModelUsage(`${targetModel.provider}/${targetModel.id}`); - - // Re-apply thinking for the newly selected model. Prefer the model's - // configured defaultLevel; otherwise preserve the current level (or auto). - this.#reapplyThinkingLevel(targetModel.thinking?.defaultLevel); - await this.#syncAfterModelChange(previousEditMode); - return { switched: true }; + return this.#models.setModel(model, role, options); } - /** - * Set model temporarily (for this session only). - * Validates that a credential source is configured (synchronously, without - * refreshing OAuth or running command-backed key programs), saves to session - * log but NOT to settings. - * @throws Error if no API key available for the model - */ - async setModelTemporary( + /** Selects a model for this session without updating persisted model settings. */ + setModelTemporary( model: Model, thinkingLevel?: ConfiguredThinkingLevel, options?: { ephemeral?: boolean }, ): Promise { - const previousEditMode = this.#resolveActiveEditMode(); - if (!this.#modelRegistry.hasConfiguredAuth(model)) { - throw new Error(`No API key for ${model.provider}/${model.id}`); - } - - const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model); - - this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); - this.#clearActiveRetryFallback(); - this.#setModelWithProviderSessionReset(targetModel); - this.sessionManager.appendModelChange( - `${targetModel.provider}/${targetModel.id}`, - options?.ephemeral ? EPHEMERAL_MODEL_CHANGE_ROLE : "temporary", - ); - this.settings.getStorage()?.recordModelUsage(`${targetModel.provider}/${targetModel.id}`); - - // Apply explicit thinking level if given; otherwise prefer the model's - // configured defaultLevel; otherwise re-clamp the current level (or auto). - if (thinkingLevel !== undefined) { - this.setThinkingLevel(thinkingLevel); - } else { - this.#reapplyThinkingLevel(targetModel.thinking?.defaultLevel); - } - await this.#syncAfterModelChange(previousEditMode); + return this.#models.setModelTemporary(model, thinkingLevel, options); } - /** - * Cycle to next/previous model. - * Uses scoped models (from --models flag) if available, otherwise all available models. - * @param direction - "forward" (default) or "backward" - * @returns The new model info, or undefined if only one model available - */ - async cycleModel(direction: "forward" | "backward" = "forward"): Promise { - if (this.#scopedModels.length > 0) { - return this.#cycleScopedModel(direction); - } - return this.#cycleAvailableModel(direction); + /** Cycles the scoped model set, or all available models when no scope exists. */ + cycleModel(direction: "forward" | "backward" = "forward"): Promise { + return this.#models.cycleModel(direction); } - /** - * Resolve the configured role models in the given order plus the index of - * the currently active one. Roles that have no configured model, or whose - * configured model is not currently available, are skipped. The `default` - * role falls back to the active model when no explicit assignment exists. - * - * Returns `undefined` only when there is no current model or no available - * models at all; an empty `models` array is never returned (callers should - * still guard on `models.length`). - */ + /** Resolves configured role models and the currently active role index. */ getRoleModelCycle(roleOrder: readonly string[]): RoleModelCycle | undefined { - const availableModels = this.#modelRegistry.getAvailable(); - if (availableModels.length === 0) return undefined; - - const currentModel = this.model; - if (!currentModel) return undefined; - const matchPreferences = getModelMatchPreferences(this.settings); - const models: ResolvedRoleModel[] = []; - - for (const role of roleOrder) { - const roleModelStr = - role === "default" - ? (this.settings.getModelRole("default") ?? `${currentModel.provider}/${currentModel.id}`) - : this.settings.getModelRole(role); - if (!roleModelStr) continue; - - const resolved = resolveModelRoleValue(roleModelStr, availableModels, { - settings: this.settings, - matchPreferences, - }); - if (!resolved.model) continue; - - models.push({ - role, - model: resolved.model, - thinkingLevel: resolved.thinkingLevel, - explicitThinkingLevel: resolved.explicitThinkingLevel, - }); - } - - if (models.length === 0) return undefined; - - // Trust the recorded role only while its resolved model still IS the - // active model. A model switch through another surface (alt+m, retry - // fallback, /model) or a role re-configuration leaves the recorded role - // pointing at a model the session no longer runs; cycling from that - // stale slot lands on the wrong neighbor and reads as a skipped entry. - const lastRole = this.sessionManager.getLastModelChangeRole(); - let currentIndex = lastRole ? models.findIndex(entry => entry.role === lastRole) : -1; - if (currentIndex !== -1 && !modelsAreEqual(models[currentIndex].model, currentModel)) { - currentIndex = -1; - } - if (currentIndex === -1) { - currentIndex = models.findIndex(entry => modelsAreEqual(entry.model, currentModel)); - } - if (currentIndex === -1) currentIndex = 0; - - return { models, currentIndex }; + return this.#models.getRoleModelCycle(roleOrder); } - /** - * Apply a resolved role model as the active model without changing global - * settings. Shared with role cycling and the plan-approval model slider. - */ - async applyRoleModel(entry: ResolvedRoleModel): Promise { - await this.setModel(entry.model, entry.role); - if (entry.explicitThinkingLevel && entry.thinkingLevel !== undefined) { - this.setThinkingLevel(entry.thinkingLevel); - } + /** Applies a resolved role model without changing global settings. */ + applyRoleModel(entry: ResolvedRoleModel): Promise { + return this.#models.applyRoleModel(entry); } - /** - * Cycle through configured role models in a fixed order. - * Skips missing roles and changes only the active session model. - * @param roleOrder - Order of roles to cycle through (e.g., ["slow", "default", "smol"]) - * @param direction - "forward" (default) or "backward" - */ - async cycleRoleModels( + /** Cycles the configured role models in the supplied order. */ + cycleRoleModels( roleOrder: readonly string[], direction: "forward" | "backward" = "forward", ): Promise { - const cycle = this.getRoleModelCycle(roleOrder); - if (!cycle || cycle.models.length <= 1) return undefined; - - const step = direction === "backward" ? -1 : 1; - const next = cycle.models[(cycle.currentIndex + step + cycle.models.length) % cycle.models.length]; - - await this.applyRoleModel(next); - - return { model: next.model, thinkingLevel: this.thinkingLevel, role: next.role }; + return this.#models.cycleRoleModels(roleOrder, direction); } - async #getScopedModelsWithApiKey(): Promise> { - const apiKeysByProvider = new Map(); - const result: Array<{ model: Model; thinkingLevel?: ThinkingLevel }> = []; - - for (const scoped of this.#scopedModels) { - const provider = scoped.model.provider; - let apiKey: string | undefined; - if (apiKeysByProvider.has(provider)) { - apiKey = apiKeysByProvider.get(provider); - } else { - apiKey = await this.#modelRegistry.getApiKeyForProvider(provider, this.sessionId); - apiKeysByProvider.set(provider, apiKey); - } - - if (apiKey) { - result.push(scoped); - } - } - - return result; - } - - async #cycleScopedModel(direction: "forward" | "backward"): Promise { - const previousEditMode = this.#resolveActiveEditMode(); - const scopedModels = await this.#getScopedModelsWithApiKey(); - if (scopedModels.length <= 1) return undefined; - - const currentModel = this.model; - let currentIndex = scopedModels.findIndex(sm => modelsAreEqual(sm.model, currentModel)); - - if (currentIndex === -1) currentIndex = 0; - const len = scopedModels.length; - const nextIndex = direction === "forward" ? (currentIndex + 1) % len : (currentIndex - 1 + len) % len; - const next = scopedModels[nextIndex]; - - // Apply model - this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(next.model)); - this.#clearActiveRetryFallback(); - this.#setModelWithProviderSessionReset(next.model); - this.sessionManager.appendModelChange(`${next.model.provider}/${next.model.id}`); - this.settings.getStorage()?.recordModelUsage(`${next.model.provider}/${next.model.id}`); - - // Apply the scoped model's configured thinking level, preserving auto. - this.setThinkingLevel(this.#autoThinking ? AUTO_THINKING : next.thinkingLevel); - await this.#syncAfterModelChange(previousEditMode); - - return { model: next.model, thinkingLevel: this.thinkingLevel, isScoped: true }; - } - - async #cycleAvailableModel(direction: "forward" | "backward"): Promise { - const previousEditMode = this.#resolveActiveEditMode(); - const availableModels = this.#modelRegistry.getAvailable(); - if (availableModels.length <= 1) return undefined; - - const currentModel = this.model; - let currentIndex = availableModels.findIndex(m => modelsAreEqual(m, currentModel)); - - if (currentIndex === -1) currentIndex = 0; - const len = availableModels.length; - const nextIndex = direction === "forward" ? (currentIndex + 1) % len : (currentIndex - 1 + len) % len; - const nextModel = availableModels[nextIndex]; - - const apiKey = await this.#modelRegistry.getApiKey(nextModel, this.sessionId); - if (!apiKey) { - throw new Error(`No API key for ${nextModel.provider}/${nextModel.id}`); - } - - this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(nextModel)); - this.#clearActiveRetryFallback(); - this.#setModelWithProviderSessionReset(nextModel); - this.sessionManager.appendModelChange(`${nextModel.provider}/${nextModel.id}`); - this.settings.getStorage()?.recordModelUsage(`${nextModel.provider}/${nextModel.id}`); - // Re-apply the current thinking level (or auto) for the newly selected model - this.#reapplyThinkingLevel(); - await this.#syncAfterModelChange(previousEditMode); - - return { model: nextModel, thinkingLevel: this.thinkingLevel, isScoped: false }; - } - - /** - * Get all available models with valid API keys, filtered by `enabledModels` when configured. - * See {@link filterAvailableModelsByEnabledPatterns} for supported pattern forms and limitations. - */ + /** Lists available models after applying the configured enabled-model filter. */ getAvailableModels(): Model[] { - const all = this.#modelRegistry.getAvailable(); - const patterns = this.settings.get("enabledModels"); - if (!patterns || patterns.length === 0) return all; - return filterAvailableModelsByEnabledPatterns(all, patterns, this.settings); + return this.#models.getAvailableModels(); } - // ========================================================================= - // Thinking Level Management - // ========================================================================= - - #applyThinkingLevelToAgent(level: ThinkingLevel | undefined): void { - this.agent.setThinkingLevel(toReasoningEffort(level)); - this.agent.setDisableReasoning(shouldDisableReasoning(level)); - } - - /** - * Set the thinking level. `auto` enables per-turn classification. Entering - * auto writes its provisional level plus `configured: "auto"` immediately, - * giving external readers an authoritative selection receipt before the next - * user turn. Later classifications persist only changed concrete resolutions. - */ + /** Selects the session thinking level and optionally persists it as the default. */ setThinkingLevel(level: ConfiguredThinkingLevel | undefined, persist: boolean = false): void { - if (level === AUTO_THINKING) { - const provisional = resolveProvisionalAutoLevel(this.model); - const wasAuto = this.#autoThinking; - const previousLevel = this.#thinkingLevel; - this.#autoThinking = true; - this.#autoResolvedLevel = undefined; - this.#thinkingLevel = provisional; - if (!wasAuto) { - this.#clearInheritedProviderPromptCacheKey(); - } - this.#applyThinkingLevelToAgent(provisional); - if (persist) { - this.settings.set("defaultThinkingLevel", AUTO_THINKING); - } - const isChanging = !wasAuto || previousLevel !== provisional; - if (isChanging) { - this.sessionManager.appendThinkingLevelChange(provisional, AUTO_THINKING); - this.#emit({ type: "thinking_level_changed", thinkingLevel: provisional, configured: AUTO_THINKING }); - } - return; - } - - const wasAuto = this.#autoThinking; - this.#autoThinking = false; - this.#autoResolvedLevel = undefined; - const effectiveLevel = resolveThinkingLevelForModel(this.model, level); - // Leaving auto must persist even when the resolved effort is unchanged (e.g. - // auto resolved to medium, then the user pins medium): otherwise the latest - // session entry keeps `configured: "auto"` and resume re-enables auto. - const isChanging = wasAuto || effectiveLevel !== this.#thinkingLevel; - - this.#thinkingLevel = effectiveLevel; - this.#applyThinkingLevelToAgent(effectiveLevel); - - if (isChanging) { - this.#clearInheritedProviderPromptCacheKey(); - this.sessionManager.appendThinkingLevelChange(effectiveLevel, effectiveLevel); - if (persist && effectiveLevel !== undefined && effectiveLevel !== ThinkingLevel.Off) { - this.settings.set("defaultThinkingLevel", effectiveLevel); - } - this.#emit({ type: "thinking_level_changed", thinkingLevel: effectiveLevel }); - } + this.#models.setThinkingLevel(level, persist); } - /** - * Re-apply the active thinking selection after a model change. Preserves `auto` - * (re-clamping the provisional level to the new model); otherwise re-applies the - * preferred default or the current effective level. - */ - #reapplyThinkingLevel(preferredDefault?: ThinkingLevel): void { - this.setThinkingLevel(this.#autoThinking ? AUTO_THINKING : (preferredDefault ?? this.#thinkingLevel)); - } - - /** - * Cycle to next thinking level: off → auto → minimal..max → off. - * @returns New selector, or undefined if model doesn't support thinking - */ + /** Advances through the thinking selectors supported by the active model. */ cycleThinkingLevel(): ConfiguredThinkingLevel | undefined { - if (!this.model?.reasoning) return undefined; - - const levels: ConfiguredThinkingLevel[] = [ - ThinkingLevel.Off, - AUTO_THINKING, - ...this.getAvailableThinkingLevels(), - ]; - const configured = this.configuredThinkingLevel(); - const currentLevel = configured === ThinkingLevel.Inherit ? ThinkingLevel.Off : configured; - const currentIndex = currentLevel ? levels.indexOf(currentLevel) : -1; - const nextIndex = (currentIndex + 1) % levels.length; - const nextLevel = levels[nextIndex]; - if (!nextLevel) return undefined; - - this.setThinkingLevel(nextLevel); - return nextLevel; + return this.#models.cycleThinkingLevel(); } - /** Timeout (ms) for per-turn auto-thinking classification before falling back. */ - static readonly #AUTO_THINKING_TIMEOUT_MS = 4000; - - /** - * Classify the current user turn and set the effective thinking level for it. - * Bounded by a timeout + abort; on any failure (no smol model, timeout, parse - * error) it falls back to the provisional concrete level and continues. Never - * throws into the turn, and never clears `#autoThinking` (auto stays active). - */ - async #applyAutoThinkingLevel(promptText: string, generation: number): Promise { - const model = this.model; - if (!model?.reasoning) return; - // Models with reasoning but no controllable effort surface (devin-agent - // Cascade routes effort via sibling model ids, not a wire param) have - // nothing to pick — skip classification rather than discard its result. - if (getSupportedEfforts(model).length === 0) return; - - let resolved: Effort | undefined; - if (this.#magicKeywordEnabled("ultrathink") && containsUltrathink(promptText)) { - // The user explicitly asked for maximum thinking; bypass the classifier - // (and its xhigh auto ceiling) and jump straight to the highest - // supported level for this model. - resolved = clampAutoThinkingEffort(model, Effort.Max); - } else { - const controller = new AbortController(); - const timer = setTimeout(() => controller.abort(), AgentSession.#AUTO_THINKING_TIMEOUT_MS); - try { - resolved = await classifyDifficulty(promptText, { - settings: this.settings, - registry: this.#modelRegistry, - model, - sessionId: this.sessionId, - signal: controller.signal, - metadataResolver: provider => this.agent.metadataForProvider(provider), - }); - } catch (error) { - logger.debug("auto-thinking: classification failed; using fallback level", { - error: error instanceof Error ? error.message : String(error), - }); - } finally { - clearTimeout(timer); - } - } - - // Drop the result if the turn was aborted/superseded while classifying. - if (this.#promptGeneration !== generation || !this.#autoThinking) return; - - const effort = resolved ?? resolveProvisionalAutoLevel(model); - if (effort === undefined) return; - const shouldPersistResolution = this.#thinkingLevel !== effort; - this.#autoResolvedLevel = effort; - this.#thinkingLevel = effort; - this.#applyThinkingLevelToAgent(effort); - if (shouldPersistResolution) { - this.sessionManager.appendThinkingLevelChange(effort, AUTO_THINKING); - } - this.#emit({ - type: "thinking_level_changed", - thinkingLevel: effort, - configured: AUTO_THINKING, - resolved: effort, - }); - } - - /** - * True when the currently selected model's family is set to `priority` — the - * `/fast` on/off state for the active model. Returns false when no model is - * selected or the model exposes no service-tier family (e.g. Fireworks, which - * has its own Providers › Fireworks Tier toggle). - * - * For "is priority actually applied to the next request?" use - * {@link isFastModeActive} instead. - */ + /** Reports whether `/fast` is enabled for the active model family. */ isFastModeEnabled(): boolean { - const family = this.model ? serviceTierFamily(this.model) : undefined; - return family ? this.#serviceTierByFamily[family] === "priority" : false; + return this.#models.isFastModeEnabled(); } - /** - * True when `priority` is actually realized on the wire for the currently - * selected model (OpenAI/Google `service_tier`, direct Anthropic fast mode, - * or Fireworks priority). Returns false for tiers the active model can't - * realize and when no model is selected. - */ + /** Reports whether priority service is realized by the active model. */ isFastModeActive(): boolean { - const model = this.model; - return !!model && realizesPriorityServiceTier(this.#effectiveServiceTier(model), model); + return this.#models.isFastModeActive(); } - /** - * Effective wire service-tier for a request to `model`. Fireworks models take - * the Priority serving path only when the Providers › Fireworks Tier setting - * is `"priority"` (and never for `-fast` variants, whose Fast serving path is - * mutually exclusive with Priority). Every other model resolves the live - * per-family tier map down to the entry for its family. - */ - #effectiveServiceTier(model: Model | undefined = this.model): ServiceTier | undefined { - if (model?.provider === "fireworks") { - return this.settings.get("providers.fireworksTier") === "priority" && !isFireworksFastModelId(model.id) - ? "priority" - : undefined; - } - if (!model) return undefined; - return resolveModelServiceTier(this.#serviceTierByFamily, model); - } - - /** The live per-family tier map, or `null` when empty (for session persistence). */ - #serviceTierEntry(): ServiceTierByFamily | null { - return Object.keys(this.#serviceTierByFamily).length > 0 ? this.#serviceTierByFamily : null; - } - - /** Set one family's tier (or clear it with `undefined`); persists the change. */ + /** Sets or clears one model family's live service tier. */ setServiceTierFamily(family: ServiceTierFamily, tier: ServiceTier | undefined): void { - if (this.#serviceTierByFamily[family] === tier) return; - const next: ServiceTierByFamily = { ...this.#serviceTierByFamily }; - if (tier) next[family] = tier; - else delete next[family]; - this.#applyServiceTierByFamily(next); + this.#models.setServiceTierFamily(family, tier); } - /** Replace the whole per-family tier map; persists + re-arms Anthropic fast mode. */ - #applyServiceTierByFamily(next: ServiceTierByFamily): void { - // Re-arming Anthropic priority clears the per-session fast-mode auto-disable - // so the next request actually carries `speed: "fast"` again. - if (next.anthropic === "priority" && this.#serviceTierByFamily.anthropic !== "priority") { - clearAnthropicFastModeFallback(this.#providerSessionState); - } - this.#serviceTierByFamily = next; - this.sessionManager.appendServiceTierChange(this.#serviceTierEntry()); - } - - /** - * `/fast on|off` targets the family of the currently selected model: it sets - * (or clears) that family's `priority` tier. Returns `false` when the model - * has no service-tier family, so callers can report that fast mode is - * unavailable instead of claiming success. - */ + /** Enables or disables priority service for the active model family. */ setFastMode(enabled: boolean): boolean { - const family = this.model ? serviceTierFamily(this.model) : undefined; - if (!family) { - this.emitNotice("info", "The current model has no service-tier control for /fast to toggle.", "priority"); - return false; - } - if (!enabled) { - if (this.#serviceTierByFamily[family] === "priority") this.setServiceTierFamily(family, undefined); - return true; - } - this.setServiceTierFamily(family, "priority"); - return true; + return this.#models.setFastMode(enabled); } + /** Toggles priority service for the active model family. */ toggleFastMode(): boolean { - if (!this.setFastMode(!this.isFastModeEnabled())) return false; - return this.isFastModeEnabled(); + return this.#models.toggleFastMode(); } - /** - * Get available thinking levels for current model. - */ + /** Lists thinking levels supported by the active model. */ getAvailableThinkingLevels(): ReadonlyArray { - if (!this.model) return []; - return getSupportedEfforts(this.model); + return this.#models.getAvailableThinkingLevels(); } // ========================================================================= @@ -11075,636 +5952,6 @@ export class AgentSession { this.settings.set("interruptMode", mode); } - // ========================================================================= - // Compaction - // ========================================================================= - - /** - * Append plan-read protection to a prune/shake config so the active plan - * file survives compaction alongside skill reads (the config defaults - * already carry skill protection). The matcher reads the current plan - * reference path at match time, so retitled plans are covered. - */ - #withPlanProtection(config: T): T { - const planMatcher = createPlanReadMatcher(() => this.#planReferencePath); - return { ...config, protectedTools: [...config.protectedTools, planMatcher] }; - } - - async #pruneToolOutputs(): Promise<{ prunedCount: number; tokensSaved: number } | undefined> { - const branchEntries = this.sessionManager.getBranch(); - const keepBoundaryId = getLatestCompactionEntry(branchEntries)?.firstKeptEntryId; - const result = pruneToolOutputs( - branchEntries, - this.#withPlanProtection({ - ...DEFAULT_PRUNE_CONFIG, - pruneUseless: this.settings.getGroup("compaction").dropUseless, - // Cache-stable boundary: never re-write the warm, already-sent prefix - // (deep stale/age victims) or summarized-away entries every turn. - keepBoundaryId, - cacheWarmSuffixTokens: PRUNE_CACHE_WARM_SUFFIX_TOKENS, - }), - ); - if (result.prunedCount === 0) { - return undefined; - } - - await this.sessionManager.rewriteEntries(); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#resetAllAdvisorRuntimes(); - this.#syncTodoPhasesFromBranch(); - this.#closeCodexProviderSessionsForHistoryRewrite(); - return result; - } - - /** - * Per-turn stale-result pass: prune older `read` results that a newer read - * of the same file has made stale, plus results their tool flagged - * contextually useless. Cache-aware (only fires when the suffix after a - * candidate is small or the session has been idle long enough that the - * provider prompt cache is cold), so it is cheap to run every turn. Gated - * on the `compaction.supersedeReads` and `compaction.dropUseless` settings. - * - * Persists via `rewriteEntries` like every other history rewrite — the - * session file must match the live (pruned) context or file-based forks - * (`/fork`, `/tan`) and resume rebuild a divergent prefix and cold-miss the - * provider prompt cache. - */ - async #pruneStaleToolResults(): Promise<{ prunedCount: number; tokensSaved: number } | undefined> { - const { supersedeReads, dropUseless } = this.settings.getGroup("compaction"); - if (!supersedeReads && !dropUseless) return undefined; - const branchEntries = this.sessionManager.getBranch(); - const keepBoundaryId = getLatestCompactionEntry(branchEntries)?.firstKeptEntryId; - const result = pruneSupersededToolResults( - branchEntries, - this.#withPlanProtection({ - supersedeKey: supersedeReads ? readToolSupersedeKey : undefined, - pruneUseless: dropUseless, - protectedTools: [...DEFAULT_PRUNE_CONFIG.protectedTools], - // Never re-write summarized-away entries; only flush the whole sent - // region once the cache is genuinely cold (idle exceeds the 1h TTL). - keepBoundaryId, - idleFlushMs: PRUNE_IDLE_FLUSH_MS, - }), - ); - if (result.prunedCount === 0) { - return undefined; - } - - await this.sessionManager.rewriteEntries(); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#resetAllAdvisorRuntimes(); - this.#syncTodoPhasesFromBranch(); - this.#closeCodexProviderSessionsForHistoryRewrite(); - return result; - } - - /** - * Strip image content blocks from every message on the current branch and - * persist the rewrite. Walks `SessionManager.getBranch()` in place — both - * `SessionMessageEntry.message` and `CustomMessageEntry.content` arrays - * are mutated, then `rewriteEntries` durably commits the new shape. The - * agent's runtime view is rebuilt from the freshly-mutated entries so any - * provider sessions caching message identity (Codex Responses) are torn - * down to force a clean replay on the next turn. - * - * No-op when the branch carries no images; returns `{ removed: 0 }` and - * skips the disk rewrite. - */ - async dropImages(): Promise<{ removed: number }> { - const branchEntries = this.sessionManager.getBranch(); - let removed = 0; - for (const entry of branchEntries) { - if (entry.type === "message") { - removed += stripImagesFromMessage(entry.message); - continue; - } - if (entry.type === "custom_message" && typeof entry.content !== "string") { - const kept: typeof entry.content = []; - let dropped = 0; - for (const part of entry.content) { - if (part.type === "image") { - dropped++; - } else { - kept.push(part); - } - } - if (dropped > 0) { - if (kept.length === 0) { - kept.push({ type: "text", text: "[image removed]" }); - } - entry.content = kept; - removed += dropped; - } - } - } - if (removed === 0) { - return { removed: 0 }; - } - await this.sessionManager.rewriteEntries(); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#resetAllAdvisorRuntimes(); - this.#closeCodexProviderSessionsForHistoryRewrite(); - return { removed }; - } - - /** - * Surgically reduce context by dropping heavy content ("shake"). - * - * - `images` delegates to {@link dropImages}. - * - `elide` replaces whole tool-call results and large fenced/XML blocks - * with short placeholders that embed an `artifact://` recovery link. - * - * Mutates the branch in place, persists via `rewriteEntries`, replays the - * rebuilt context through the agent, and tears down provider sessions that - * cache message identity — same rewrite contract as {@link dropImages}. - * - * No-op (zero counts) when nothing is eligible. - */ - async shake(mode: ShakeMode, opts: { config?: ShakeConfig; signal?: AbortSignal } = {}): Promise { - if (mode === "images") { - const { removed } = await this.dropImages(); - return { mode, toolResultsDropped: 0, blocksDropped: 0, imagesDropped: removed, tokensFreed: 0 }; - } - - const branchEntries = this.sessionManager.getBranch(); - const config = this.#withPlanProtection({ - ...(opts.config ?? AGGRESSIVE_SHAKE_CONFIG), - // Skip entries summarized away by the latest compaction — shaking them - // only churns persisted history with no prompt/cache effect. - keepBoundaryId: getLatestCompactionEntry(branchEntries)?.firstKeptEntryId, - }); - const regions = collectShakeRegions(branchEntries, config); - if (regions.length === 0) { - return { mode, toolResultsDropped: 0, blocksDropped: 0, tokensFreed: 0 }; - } - - const artifactId = await this.#saveShakeArtifact(regions); - const replacements = regions.map((region, index) => this.#shakeElidePlaceholder(region, index, artifactId)); - - let toolResultsDropped = 0; - let blocksDropped = 0; - let originalTokens = 0; - let replacementTokens = 0; - const items = regions.map((region, index) => { - if (region.kind === "toolResult") toolResultsDropped++; - else blocksDropped++; - originalTokens += region.tokens; - const replacement = replacements[index]; - if (replacement.length > 0) replacementTokens += countTokens(replacement); - return { region, replacement }; - }); - - applyShakeRegions(items); - - await this.sessionManager.rewriteEntries(); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#resetAllAdvisorRuntimes(); - this.#closeCodexProviderSessionsForHistoryRewrite(); - - return { - mode, - toolResultsDropped, - blocksDropped, - tokensFreed: Math.max(0, originalTokens - replacementTokens), - artifactId, - }; - } - - #shakeElidePlaceholder(region: ShakeRegion, index: number, artifactId: string | undefined): string { - if (artifactId) { - return `[shaken ~${region.tokens} tokens — recover: artifact://${artifactId} (region ${index + 1})]`; - } - return `[shaken ~${region.tokens} tokens]`; - } - - /** - * Concatenate the original region contents into one session artifact so the - * agent can read them back via `artifact://`. Returns `undefined` when - * the session is not persisted or the write fails — callers degrade to a - * bare placeholder. - */ - async #saveShakeArtifact(regions: ShakeRegion[]): Promise { - const parts: string[] = []; - for (let i = 0; i < regions.length; i++) { - const region = regions[i]; - parts.push(`### region ${i + 1} (${region.label}, ~${region.tokens} tok)`, "", region.originalText, ""); - } - try { - return await this.sessionManager.saveArtifact(parts.join("\n"), "shake"); - } catch { - return undefined; - } - } - - /** - * Manually compact the session context. - * Aborts current agent operation first. - * @param customInstructions Optional instructions for the compaction summary - * @param options Optional callbacks for completion/error handling - */ - async compact(customInstructions?: string, options?: CompactOptions): Promise { - if (this.#compactionAbortController) { - throw new Error("Compaction already in progress"); - } - // Resolve the `/compact ` subcommand up front so input validation - // runs before we disconnect/abort the active agent operation below. - const compactMode = options?.mode ? findCompactMode(options.mode) : undefined; - // Modes that produce no LLM summary (snapcompact) have nothing to focus. - // Reject focus text loudly so programmatic callers don't silently lose - // instructions (the slash path pre-validates via parseCompactArgs). - // `internalGuidance` counts the same way — plan-mode approval never - // combines with a rejects-focus mode, but reject early if a caller ever - // wires it up so we don't silently drop the directive on the snapcompact - // fallback (issue #4359). - if (compactMode?.rejectsFocus && (customInstructions || options?.internalGuidance)) { - throw new Error(`/compact ${compactMode.name} does not take focus instructions.`); - } - const compactionAbortController = new AbortController(); - this.#compactionAbortController = compactionAbortController; - - try { - this.#disconnectFromAgent(); - await this.abort({ goalReason: "internal", preserveCompaction: true }); - if (!this.model) { - throw new Error("No model selected"); - } - - const compactionSettings = this.settings.getGroup("compaction"); - // The `/compact ` override (resolved above) replaces the configured - // strategy/remote flags for this one invocation. Merged before - // prepareCompaction so the remote gating (preparation.settings. - // remoteEnabled/endpoint) and the snapcompact decision below both see it. - const effectiveSettings = compactMode - ? { ...compactionSettings, ...compactMode.overrides } - : compactionSettings; - // /compact remote demands provider-native compaction. When no remote - // endpoint is configured (one would override per-model gating in - // compact()), drop fallback candidates that aren't remote-capable so the - // engine never silently runs a local summary on a configured-but-non- - // remote compactionModel. If filtering empties the chain, warn and fall - // back to the full chain so the operation still completes. - const availableModels = this.#modelRegistry.getAvailable(); - const requireProviderRemote = Boolean(compactMode?.requiresRemote && !effectiveSettings.remoteEndpoint); - let compactionCandidates = this.#getCompactionModelCandidates( - availableModels, - requireProviderRemote ? shouldUseOpenAiRemoteCompaction : undefined, - ); - if (requireProviderRemote && compactionCandidates.length === 0) { - this.emitNotice( - "warning", - `remote compaction is unavailable for ${this.model.id} (no remote endpoint configured and no provider-native remote-capable model in the fallback chain) — using a local summary instead`, - "compaction", - ); - compactionCandidates = this.#getCompactionModelCandidates(availableModels); - } - const pathEntries = this.sessionManager.getBranch(); - const preparation = prepareCompaction(pathEntries, effectiveSettings, this.model); - if (!preparation) { - // Check why we can't compact - const lastEntry = pathEntries[pathEntries.length - 1]; - if (lastEntry?.type === "compaction") { - throw new Error("Already compacted"); - } - throw new Error("Nothing to compact (session too small)"); - } - - let hookCompaction: CompactionResult | undefined; - let fromExtension = false; - let preserveData: Record | undefined; - - if (this.#extensionRunner?.hasHandlers("session_before_compact")) { - const result = (await this.#extensionRunner.emit({ - type: "session_before_compact", - preparation, - branchEntries: pathEntries, - customInstructions, - signal: compactionAbortController.signal, - })) as SessionBeforeCompactResult | undefined; - - if (result?.cancel) { - throw new CompactionCancelledError(); - } - - if (result?.compaction) { - hookCompaction = result.compaction; - fromExtension = true; - } - } - - const compactionPrep = await this.#prepareCompactionFromHooks(preparation, hookCompaction); - - // Strategy honored on manual /compact too. Custom instructions (public - // user focus OR internal plan-mode guidance) imply a directed LLM - // summary; a text-only model cannot read snapcompact frames. - const wantsSnapcompact = - compactionPrep.kind !== "fromHook" && - effectiveSettings.strategy === "snapcompact" && - !customInstructions && - !options?.internalGuidance; - // `/compact snapcompact` is an explicit no-LLM archive request: honor - // its contract by failing locally rather than silently shipping the - // transcript to a provider. The default-configured snapcompact - // strategy, in contrast, falls back to LLM compaction (mirroring the - // auto-compaction path) so a routine /compact still completes on a - // text-only model (issue #5064). - const explicitSnapcompact = compactMode?.name === "snapcompact"; - let snapcompactReady = wantsSnapcompact; - const snapcompactShapeSetting = this.settings.get("snapcompact.shape"); - let snapcompactShape: snapcompact.Shape | undefined; - // Claude refuses inputs that reproduce its own reasoning as text - // ("reasoning_extraction"), and the snapcompact archive is replayed as - // text into every later request; drop `¶think:` sections for - // Anthropic-dialect targets (issue #6093). - const snapcompactIncludeThinking = preferredDialect(this.model.id) !== "anthropic"; - if (wantsSnapcompact && !this.model.input.includes("image")) { - if (explicitSnapcompact) { - this.emitNotice( - "warning", - `snapcompact needs a vision-capable model (${this.model.id} is text-only)`, - "compaction", - ); - throw new Error(`snapcompact cannot run locally: ${this.model.id} is text-only.`); - } - this.emitNotice( - "warning", - `snapcompact needs a vision-capable model (${this.model.id} is text-only); falling back to LLM compaction`, - "compaction", - ); - snapcompactReady = false; - } else if (snapcompactReady) { - const text = snapcompact.serializeConversation( - convertToLlm(preparation.messagesToSummarize.concat(preparation.turnPrefixMessages)), - { includeThinking: snapcompactIncludeThinking }, - ); - const probeText = snapcompact.renderabilityProbeText( - text, - preparation.previousPreserveData, - preparation.previousSummary, - ); - snapcompactShape = snapcompact.resolveShapeForText(probeText, this.model, snapcompactShapeSetting); - const renderScan = snapcompact.scanRenderability(probeText, { shape: snapcompactShape }); - if (!renderScan.isSafe) { - const percent = (renderScan.unrenderableRatio * 100).toFixed(1); - this.emitNotice( - "warning", - `snapcompact disabled: unsupported characters for selected snapcompact font (${percent}%). No LLM fallback was attempted.`, - "compaction", - ); - throw new Error( - `snapcompact cannot render this conversation locally: unsupported characters for selected snapcompact font (${percent}%).`, - ); - } - } - - let summary: string; - let shortSummary: string | undefined; - let firstKeptEntryId: string; - let tokensBefore: number; - let details: unknown; - let codexCompaction: CodexCompactionContext | undefined; - - // Snapcompact runs locally first. The frame cap is sized from the live - // model window via #computeSnapcompactMaxFrames so the post-render context - // fits without the warning loop (issue #3247). Zero-frame budget now fails - // the snapcompact request locally rather than falling back to an LLM call. - let snapcompactResult: snapcompact.CompactionResult | undefined; - if (snapcompactReady) { - const maxFrames = this.#computeSnapcompactMaxFrames(preparation, effectiveSettings); - if (maxFrames < 1) { - logger.warn("Snapcompact skipped: kept history alone exceeds the context budget", { - model: this.model?.id, - }); - this.emitNotice( - "warning", - "snapcompact: kept history alone exceeds the context budget. No LLM fallback was attempted.", - "compaction", - ); - throw new Error("snapcompact cannot run locally: kept history alone exceeds the context budget."); - } else { - const shape = snapcompactShape; - if (!shape) { - throw new Error("snapcompact shape was not resolved before rendering."); - } - snapcompactResult = await snapcompact.compact(preparation, { - convertToLlm, - model: this.model, - ...(snapcompactShapeSetting === "auto" ? {} : { shape }), - maxFrames, - includeThinking: snapcompactIncludeThinking, - }); - const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); - if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { - logger.warn("Snapcompact exceeded the per-request frame payload budget", { - model: this.model?.id, - framePayloadBytes, - budget: snapcompact.FRAME_DATA_BYTES_BUDGET, - }); - this.emitNotice( - "warning", - "snapcompact produced too much standing image payload. No LLM fallback was attempted.", - "compaction", - ); - throw new Error( - "snapcompact cannot run locally: standing image payload exceeds the per-request budget.", - ); - } - const ctxWindow = this.model?.contextWindow ?? 0; - const budget = - ctxWindow > 0 - ? ctxWindow - effectiveReserveTokens(ctxWindow, effectiveSettings) - : Number.POSITIVE_INFINITY; - if (this.#projectSnapcompactContextTokens(preparation, snapcompactResult) > budget) { - logger.warn("Snapcompact still overflows the window after frame-budget sizing", { - model: this.model?.id, - }); - this.emitNotice( - "warning", - "snapcompact could not bring the context under the limit. No LLM fallback was attempted.", - "compaction", - ); - throw new Error("snapcompact could not bring the context under the limit locally."); - } - } - } - - if (compactionPrep.kind === "fromHook") { - summary = compactionPrep.summary; - shortSummary = compactionPrep.shortSummary; - firstKeptEntryId = compactionPrep.firstKeptEntryId; - tokensBefore = compactionPrep.tokensBefore; - details = compactionPrep.details; - preserveData = compactionPrep.preserveData; - } else if (snapcompactResult) { - summary = snapcompactResult.summary; - shortSummary = snapcompactResult.shortSummary; - firstKeptEntryId = snapcompactResult.firstKeptEntryId; - tokensBefore = snapcompactResult.tokensBefore; - details = snapcompactResult.details; - preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) }; - } else { - codexCompaction = createCodexCompactionContext({ - trigger: "manual", - reason: "user_requested", - phase: "standalone_turn", - }); - // Generate compaction result. Only convert known abort-shaped - // rejections (AbortError raised while the abort signal is set, - // or an already-typed sentinel) into `CompactionCancelledError` - // so downstream callers can discriminate cancel from generic - // failure via `instanceof` without inspecting message strings. - // Real compaction bugs (network, server, parsing, etc.) keep - // their original shape — they must not be silently relabeled - // as cancellations even if the signal happens to be aborted - // for an unrelated reason. Assignments live inside the try - // block because every catch path throws — the post-try reads - // of the result-derived locals are reachable only on success. - try { - const result = await this.#compactWithFallbackModel( - preparation, - options?.internalGuidance ?? customInstructions, - compactionAbortController.signal, - { - promptOverride: this.#obfuscateTextForProvider(compactionPrep.hookPrompt), - extraContext: compactionPrep.hookContext, - remoteInstructions: this.#baseSystemPrompt.join("\n\n"), - convertToLlm: messages => this.#convertToLlmForSideRequest(messages), - codexCompaction, - }, - compactionCandidates, - ); - summary = result.summary; - shortSummary = result.shortSummary; - firstKeptEntryId = result.firstKeptEntryId; - tokensBefore = result.tokensBefore; - details = result.details; - preserveData = mergeLlmCompactionPreserveData(compactionPrep.preserveData, result.preserveData); - } catch (err) { - if (err instanceof CompactionCancelledError) { - throw err; - } - if (compactionAbortController.signal.aborted && err instanceof Error && err.name === "AbortError") { - throw new CompactionCancelledError(); - } - throw err; - } - } - - if (compactionAbortController.signal.aborted) { - throw new CompactionCancelledError(); - } - - this.sessionManager.appendCompaction( - summary, - shortSummary, - firstKeptEntryId, - tokensBefore, - details, - fromExtension, - preserveData, - ); - const newEntries = this.sessionManager.getEntries(); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#rebasePendingContextSnapshotAfterCompaction(); - // Compaction discarded the conversation history that carried the approved - // plan reference. Clear the sent-flag so #buildPlanReferenceMessage re-reads - // the plan from disk and re-injects it on the next turn (issue #1246). - this.#planReferenceSent = false; - this.#resetAllAdvisorRuntimes(); - this.#syncTodoPhasesFromBranch(); - if (codexCompaction) { - this.#resetCodexProviderAfterCompaction(codexCompaction); - } else { - this.#closeCodexProviderSessionsForHistoryRewrite(); - } - - // Get the saved compaction entry for the hook - const savedCompactionEntry = newEntries.find(e => e.type === "compaction" && e.summary === summary) as - | CompactionEntry - | undefined; - - if (this.#extensionRunner && savedCompactionEntry) { - await this.#extensionRunner.emit({ - type: "session_compact", - compactionEntry: savedCompactionEntry, - fromExtension, - }); - } - - const compactionResult: CompactionResult = { - summary, - shortSummary, - firstKeptEntryId, - tokensBefore, - details, - preserveData, - }; - options?.onComplete?.(compactionResult); - return compactionResult; - } catch (error) { - const err = error instanceof Error ? error : new Error(String(error)); - options?.onError?.(err); - throw error; - } finally { - if (this.#compactionAbortController === compactionAbortController) { - this.#compactionAbortController = undefined; - } - this.#reconnectToAgent(); - // Compaction disconnected before `await abort()`, so abort's finally drain - // (and any steer/follow-up that arrived mid-compaction — async IRC, an - // `xd://` mount notice, an SDK/RPC steer) was suppressed while disconnected - // (issue #5800). Unlike `/new`/switchSession, compaction preserves the agent - // queues, so nothing else resumes them: re-drain now that the listener is back - // and `isCompacting` is false, or the queued turn hangs until the next prompt. - this.#drainStrandedQueuedMessages(); - } - } - - /** - * Ask the active memory backend for an extra-context block to splice into - * the compaction summary prompt. Both the manual and auto compaction paths - * funnel through this helper so the behaviour stays identical. - * - * Failures are swallowed: a memory backend going sideways MUST NOT block - * compaction (which is itself the recovery path for context overflow). - */ - async #collectMemoryBackendContext(preparation: { - messagesToSummarize: AgentMessage[]; - turnPrefixMessages: AgentMessage[]; - }): Promise { - const backend = await resolveMemoryBackend(this.settings); - if (!backend.preCompactionContext) return undefined; - const messages = preparation.messagesToSummarize.concat(preparation.turnPrefixMessages); - try { - return await backend.preCompactionContext(messages, this.settings, this); - } catch (err) { - logger.debug("Memory backend preCompactionContext failed", { - backend: backend.id, - error: String(err), - }); - return undefined; - } - } - - /** - * Cancel in-progress context maintenance (manual compaction, auto-compaction, or auto-handoff). - */ - abortCompaction(): void { - this.#compactionAbortController?.abort(); - this.#autoCompactionAbortController?.abort(); - this.#handoffAbortController?.abort(); - } - - /** Trigger idle compaction through the auto-compaction flow (with UI events). */ - async runIdleCompaction(): Promise { - if (this.isStreaming || this.isCompacting) return; - await this.#runAutoCompaction("idle", false, true); - } - /** * Cancel in-progress branch summarization. */ @@ -11716,14 +5963,14 @@ export class AgentSession { * Cancel in-progress handoff generation. */ abortHandoff(): void { - this.#handoffAbortController?.abort(); + this.#handoff.abortHandoff(); } /** * Check if handoff generation is in progress. */ get isGeneratingHandoff(): boolean { - return this.#handoffAbortController !== undefined; + return this.#handoff.isGeneratingHandoff; } /** @@ -11733,612 +5980,10 @@ export class AgentSession { * @param options Handoff execution options * @returns The handoff document text, or undefined if cancelled/failed */ - async handoff(customInstructions?: string, options?: SessionHandoffOptions): Promise { - this.#assertVibeSessionTransitionAllowed("handoff to a new session"); - const entries = this.sessionManager.getBranch(); - const messageCount = entries.filter(e => e.type === "message").length; - - if (messageCount < 2) { - throw new Error("Nothing to hand off (no messages yet)"); - } - - this.#skipPostTurnMaintenanceAssistantTimestamp = undefined; - - this.#handoffAbortController = new AbortController(); - const handoffAbortController = this.#handoffAbortController; - const handoffSignal = handoffAbortController.signal; - const sourceSignal = options?.signal; - const onSourceAbort = () => { - if (!handoffSignal.aborted) { - handoffAbortController.abort(); - } - }; - if (sourceSignal) { - sourceSignal.addEventListener("abort", onSourceAbort, { once: true }); - if (sourceSignal.aborted) { - onSourceAbort(); - } - } - - try { - if (handoffSignal.aborted) { - throw new Error("Handoff cancelled"); - } - - const model = this.model; - if (!model) { - throw new Error("No model selected for handoff"); - } - const apiKey = await this.#modelRegistry.getApiKey(model, this.sessionId); - if (!apiKey) { - throw new Error(`No API key for ${model.provider}`); - } - - // Build the handoff request through the SAME pipeline a live turn uses - // (`runEphemeralTurn` / `/btw` share it) so the oneshot reads the - // provider prompt cache the main turn populated instead of cold-missing - // the whole prefix: identical system prompt, normalized tools, and - // transform-/obfuscation-matched message history via - // `convertMessagesToLlm` + `buildSideRequestContext`, plus the live turn's - // effective provider cache key with a unique side `sessionId` so - // OpenAI/Codex append-only state never mixes with the live turn. - const cacheSessionId = this.sessionId; - // The loop sends `promptCacheKey` (providerPromptCacheKey) and falls back to - // the provider session id; providers route on `promptCacheKey ?? sessionId`. - // Both can diverge from this.sessionId (tan/subagent/shared sessions), so - // mirror exactly what the live turn populated the cache under. - const handoffPromptCacheKey = this.agent.promptCacheKey ?? this.agent.sessionId; - const handoffPromptText = renderHandoffPrompt(this.#obfuscateTextForProvider(customInstructions)); - const handoffSnapshot: AgentMessage[] = [ - ...this.agent.state.messages, - { - role: "user", - content: [{ type: "text", text: handoffPromptText }], - attribution: "agent", - timestamp: Date.now(), - }, - ]; - const handoffLlmMessages = await this.convertMessagesToLlm(handoffSnapshot, handoffSignal); - // Base system prompt, not a per-turn `before_agent_start` hook override — - // the handoff seeds a fresh session and must not carry prompt-specific - // hook state. Matches the prompt the old handoff path sent. - const handoffContext = await this.agent.buildSideRequestContext(handoffLlmMessages, this.#baseSystemPrompt); - const handoffStreamOptions = this.prepareSimpleStreamOptions( - { - apiKey: this.#modelRegistry.resolver(model, cacheSessionId), - sessionId: `${cacheSessionId}:side:${Snowflake.next()}`, - promptCacheKey: handoffPromptCacheKey, - preferWebsockets: false, - serviceTier: this.#effectiveServiceTier(model), - hideThinkingSummary: this.agent.hideThinkingSummary, - initiatorOverride: "agent", - signal: handoffSignal, - }, - model.provider, - ); - const rawHandoffText = await generateHandoffFromContext( - obfuscateProviderContext(this.#obfuscator, handoffContext), - model, - { - streamOptions: handoffStreamOptions, - completeImpl: async (requestModel, requestContext, requestOptions) => { - const stream = await this.#sideStreamFn(requestModel, requestContext, requestOptions); - return stream.result(); - }, - telemetry: resolveTelemetry(this.agent.telemetry, this.sessionId), - // Honor the user's /model thinking selection on the handoff path. - // Clamped per-model inside generateHandoffFromContext via - // resolveCompactionEffort so unsupported-effort models don't trip - // requireSupportedEffort. - thinkingLevel: this.thinkingLevel, - }, - ); - const handoffText = this.#deobfuscateFromProvider(rawHandoffText); - - if (handoffSignal.aborted) { - throw new Error("Handoff cancelled"); - } - if (!handoffText) { - return undefined; - } - - // Start a new session - const previousSessionFile = this.sessionFile; - if (this.#extensionRunner?.hasHandlers("session_before_switch")) { - const result = (await this.#extensionRunner.emit({ - type: "session_before_switch", - reason: "handoff", - })) as SessionBeforeSwitchResult | undefined; - - if (result?.cancel) { - options?.onSwitchCancelled?.(); - return undefined; - } - } - await this.#flushPendingBashMessages(); - await this.sessionManager.flush(); - const bashTransition = this.#beginBashSessionTransition(); - this.#cancelOwnAsyncJobs(); - let sessionTransitioned = false; - try { - await this.sessionManager.newSession( - previousSessionFile ? { parentSession: previousSessionFile } : undefined, - ); - this.#markBashSessionTransition(bashTransition); - sessionTransitioned = true; - } finally { - this.#finishBashSessionTransition(bashTransition, sessionTransitioned); - } - - this.#clearCheckpointRuntimeState(); - // agent.reset() clears the core steering/follow-up queues. Preserve any queued - // steers/follow-ups (RPC/SDK steer()/followUp() issued during the handoff, or a - // pre-loader TUI steer) so they survive into the post-handoff session instead of - // being silently dropped. Capture is synchronous immediately before reset and - // restore is synchronous immediately after — no await gap — so a steer arriving - // later (during ensureOnDisk/Bun.write below) appends to the restored queue - // rather than being clobbered. - const preservedSteering = this.agent.peekSteeringQueue().slice(); - const preservedFollowUp = this.agent.peekFollowUpQueue().slice(); - this.agent.reset(); - this.agent.replaceQueues(preservedSteering, preservedFollowUp); - this.#freshProviderSessionId = undefined; - this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); - await this.#resetMemoryContextForNewTranscript(); - this.#pendingNextTurnMessages = []; - this.#scheduledHiddenNextTurnGeneration = undefined; - this.#todoReminderCount = 0; - this.#todoReminderAwaitingProgress = false; - this.#mutationsSinceLastTodoTouch = 0; - this.#midRunNudgeCount = 0; - - // Inject the handoff document as a custom message - const handoffContent = createHandoffContext(handoffText); - this.sessionManager.appendCustomMessageEntry("handoff", handoffContent, true, undefined, "agent"); - await this.sessionManager.ensureOnDisk(); - let savedPath: string | undefined; - if (options?.autoTriggered && this.settings.get("compaction.handoffSaveToDisk")) { - const artifactsDir = this.sessionManager.getArtifactsDir(); - if (artifactsDir) { - const handoffFilePath = path.join(artifactsDir, createHandoffFileName()); - try { - await Bun.write(handoffFilePath, `${handoffText}\n`); - savedPath = handoffFilePath; - } catch (error) { - logger.warn("Failed to save handoff document to disk", { - path: handoffFilePath, - error: error instanceof Error ? error.message : String(error), - }); - } - } else { - logger.debug("Skipping handoff document save because session is not persisted"); - } - } - - // Rebuild agent messages from session - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#resetAllAdvisorRuntimes(); - this.#syncTodoPhasesFromBranch(); - if (this.#extensionRunner) { - await this.#extensionRunner.emit({ - type: "session_switch", - reason: "handoff", - previousSessionFile, - }); - } - - return { document: handoffText, savedPath }; - } catch (error) { - if (handoffSignal.aborted || (error instanceof Error && error.name === "AbortError")) { - throw new Error("Handoff cancelled"); - } - throw error; - } finally { - sourceSignal?.removeEventListener("abort", onSourceAbort); - this.#handoffAbortController = undefined; - } + handoff(customInstructions?: string, options?: SessionHandoffOptions): Promise { + return this.#handoff.handoff(customInstructions, options); } - /** - * Local token estimate of the stored conversation (plus any pending messages), - * independent of provider-reported usage. A `before_provider_request` hook - * (e.g. a compression extension such as Headroom) or other on-wire payload - * transform can shrink the request below the real stored conversation; the - * provider then reports deflated prompt tokens, so anchoring the compaction - * decision purely on that usage lets the real history grow unbounded until it - * overflows and native compaction can no longer run. This estimate is the - * floor the compaction decision respects so on-wire compression can never - * suppress it. - */ - #estimateStoredContextTokens(pendingMessages: AgentMessage[] = []): number { - // Exclude encrypted reasoning (thinkingSignature / redactedThinking): its - // local byte size diverges from what the provider bills, so counting it here - // would let a thinking-heavy turn falsely trip the floor. The provider usage - // (the other arm of compactionContextTokens) already accounts for it. - const opts = { excludeEncryptedReasoning: true } as const; - return ( - computeNonMessageTokens(this) + - this.messages.reduce((sum, msg) => sum + estimateTokens(msg, opts), 0) + - pendingMessages.reduce((sum, msg) => sum + estimateTokens(msg, opts), 0) - ); - } - - #estimatePrePromptContextTokens(messages: AgentMessage[], contextWindow: number): number { - const breakdown = this.getContextBreakdown({ contextWindow, pendingMessages: messages }); - const localEstimate = this.#estimateStoredContextTokens(messages); - // Floor by the local estimate: a payload-shrinking before_provider_request - // hook deflates the provider-anchored breakdown, which must not suppress - // pre-prompt compaction (see #estimateStoredContextTokens). - return compactionContextTokens(breakdown?.usedTokens ?? 0, localEstimate); - } - - async #runPrePromptCompactionIfNeeded(messages: AgentMessage[]): Promise { - const model = this.model; - if (!model) return; - const contextWindow = model.contextWindow ?? 0; - if (contextWindow <= 0) return; - const compactionSettings = this.settings.getGroup("compaction"); - const contextTokens = this.#estimatePrePromptContextTokens(messages, contextWindow); - if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) return; - - // Auto-promote first: switching to a larger-context model avoids compacting - // the history at all. The post-turn threshold path already promotes before - // compacting; without this, the pre-prompt path would pre-empt promotion and - // compact (snapcompact/summary) a session that should have just been promoted. - if (await this.#promoteContextModel()) { - logger.debug("Pre-prompt context promotion avoided compaction", { - contextTokens, - contextWindow, - model: `${model.provider}/${model.id}`, - }); - return; - } - - logger.debug("Pre-prompt context maintenance triggered by pending prompt size", { - contextTokens, - contextWindow, - model: `${model.provider}/${model.id}`, - }); - await this.#runAutoCompaction("threshold", false, false, false, { - autoContinue: false, - triggerContextTokens: contextTokens, - phase: "pre_turn", - }); - } - - /** - * Compact continuing tool-loop runs before the next provider request. - * - * `onTurnEnd` is the safe boundary: tool results for the just-finished turn - * are already paired in `activeMessages`, the live array the agent loop reads - * before its next model call. Before compacting, the just-finished turn is - * synchronously persisted if async message hooks have not reached the normal - * append path yet. Mid-run handoff is suppressed because resetting the session - * while the loop owns `activeMessages` would race the next request; handoff - * strategy falls back to in-place context-full compaction here. - */ - async #maintainContextMidRun( - activeMessages: AgentMessage[], - signal: AbortSignal | undefined, - context: AgentTurnEndContext | undefined, - ): Promise { - if ( - signal?.aborted || - this.#isDisposed || - this.isCompacting || - this.isGeneratingHandoff || - !context?.willContinue - ) - return; - - const model = this.model; - const contextWindow = model?.contextWindow ?? 0; - if (contextWindow <= 0) return; - - const compactionSettings = this.settings.getGroup("compaction"); - if ( - !compactionSettings.enabled || - compactionSettings.strategy === "off" || - compactionSettings.midTurnEnabled === false - ) { - return; - } - - const lastAssistant = [...activeMessages] - .reverse() - .find((message): message is AssistantMessage => message.role === "assistant"); - if (!lastAssistant || lastAssistant.stopReason === "aborted" || lastAssistant.stopReason === "error") return; - - if (!(await this.#persistTurnMessagesForMidRunCompaction(context))) return; - - const billedContextTokens = calculateContextTokens(lastAssistant.usage); - const storedContextTokens = this.#estimateStoredContextTokens(); - const contextTokens = compactionContextTokens(billedContextTokens, storedContextTokens); - if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) return; - - // Promote to a larger-context sibling before compacting, mirroring the - // pre-prompt (#runPrePromptCompactionIfNeeded) and post-turn threshold - // (#checkCompaction) paths. Without this, a long mid-turn tool loop that - // crosses the threshold compacts the history (and can hit the no-progress - // dead-end on a single oversized turn) on a model that should have just - // been promoted to a larger window instead. - if (await this.#promoteContextModel()) { - logger.debug("Mid-run context promotion avoided compaction", { - contextTokens, - contextWindow, - from: `${model?.provider}/${model?.id}`, - }); - return; - } - - const messagesBefore = activeMessages.length; - await this.#runAutoCompaction("threshold", false, false, false, { - autoContinue: false, - suppressContinuation: true, - suppressHandoff: true, - triggerContextTokens: contextTokens, - phase: "mid_turn", - }); - - if (signal?.aborted) return; - const compactedMessages = this.agent.state.messages; - if (compactedMessages !== activeMessages) { - activeMessages.splice(0, activeMessages.length, ...compactedMessages); - } - logger.debug("Mid-run compaction ran between provider calls", { - contextTokens, - contextWindow, - strategy: compactionSettings.strategy, - goalActive: this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active", - messagesBefore, - messagesAfter: activeMessages.length, - }); - } - /** - * Check if context maintenance or promotion is needed and run it. - * Called after agent_end and before prompt submission. - * - * Four cases (in order): - * 1. Input overflow + promotion: promote to larger model, retry without maintenance. - * 2. Input overflow + no promotion target: run context maintenance, auto-retry on same model. - * 3. Output incomplete (stopReason === "length", e.g. `response.incomplete`): the - * model burned its output budget without producing an actionable deliverable - * (reasoning-only or truncated). Drop the dead turn, try promotion, otherwise - * run compaction/handoff and retry. - * 4. Threshold: context over threshold, run context maintenance (no auto-retry). - * - * @param assistantMessage The assistant message to check - * @param skipAbortedCheck If false, include aborted messages (for pre-prompt check). Default: true - * @param allowDefer If true, threshold-driven handoff strategy may schedule itself as a - * deferred post-prompt task instead of running inline. Callers running inside the - * `agent_end` handler set this to true so `session.prompt()` resolves cleanly; callers - * on the pre-prompt path (where the next agent turn is about to start) set it to false - * to avoid racing the deferred handoff against the new turn. - * @param autoContinue Whether maintenance may schedule the agent-authored continuation prompt. - * @returns whether compaction/recovery scheduled a handoff, retry, auto-continue, or - * queued-message drain that already owns the next turn. Callers MUST skip - * `session_stop` and other agent continuations when `continuationScheduled` - * is true. - */ - async #checkCompaction( - assistantMessage: AssistantMessage, - skipAbortedCheck = true, - allowDefer = true, - autoContinue = true, - ): Promise { - // Skip if message was aborted (user cancelled) - unless skipAbortedCheck is false - if (skipAbortedCheck && assistantMessage.stopReason === "aborted") return COMPACTION_CHECK_NONE; - const contextWindow = this.model?.contextWindow ?? 0; - const generation = this.#promptGeneration; - // Skip overflow check if the message came from a different model. - // This handles the case where user switched from a smaller-context model (e.g. opus) - // to a larger-context model (e.g. codex) - the overflow error from the old model - // shouldn't trigger compaction for the new model. - const sameModel = - this.model && assistantMessage.provider === this.model.provider && assistantMessage.model === this.model.id; - // This handles the case where an error was kept after compaction (in the "kept" region). - // The error shouldn't trigger another compaction since we already compacted. - // Example: opus fails -> switch to codex -> compact -> switch back to opus -> opus error - // is still in context but shouldn't trigger compaction again. - const compactionEntry = getLatestCompactionEntry(this.sessionManager.getBranch()); - const errorIsFromBeforeCompaction = - compactionEntry !== null && assistantMessage.timestamp < new Date(compactionEntry.timestamp).getTime(); - if (sameModel && !errorIsFromBeforeCompaction && AIError.isContextOverflow(assistantMessage, contextWindow)) { - // Clear the failed turn from active context so the retry (or the next - // user prompt) does not replay it. The persisted branch entry stays - // for now: when no recovery path runs, the user-facing transcript - // MUST keep the only assistant message explaining why the turn - // stopped. The branch entry is dropped further down, but only on the - // paths that actually schedule a retry/compaction. - this.#removeAssistantMessageFromActiveContext(assistantMessage); - - // Try context promotion first - switch to a larger model and retry without compacting - const promoted = await this.#tryContextPromotion(assistantMessage); - if (promoted) { - await this.#dropPersistedAssistantTurn(assistantMessage); - // Retry on the promoted (larger) model without compacting - this.#scheduleAgentContinue({ delayMs: 100, generation }); - return COMPACTION_CHECK_CONTINUATION; - } - - // No promotion target available fall through to compaction - const compactionSettings = this.settings.getGroup("compaction"); - if (compactionSettings.enabled && compactionSettings.strategy !== "off") { - return await this.#runRecoveryCompactionWithRollback("overflow", assistantMessage, allowDefer, { - autoContinue, - }); - } - return COMPACTION_CHECK_NONE; - } - // A context promotion can land while the failing call is already in - // flight (or on a run whose loop predates the switch): the overflow - // error then arrives stamped with the pre-promotion model while - // `this.model` is already the promoted target. The sameModel guard - // above deliberately ignores stale foreign-model errors, but this - // state is not stale — recover exactly like the promotion path: - // drop the dead turn and retry on the already-promoted model. Gated - // narrowly on "current model IS the failed model's promotion target - // with a strictly larger window" so genuinely stale errors from - // old user-switched models keep surfacing untouched. - if ( - !sameModel && - autoContinue && - !errorIsFromBeforeCompaction && - assistantMessage.stopReason === "error" && - this.model && - contextWindow > 0 && - this.settings.getGroup("contextPromotion").enabled - ) { - const failedModel = this.#modelRegistry.find(assistantMessage.provider, assistantMessage.model); - const failedWindow = failedModel?.contextWindow ?? 0; - const promotionTarget = failedModel - ? this.#resolveContextPromotionConfiguredTarget(failedModel, this.#modelRegistry.getAvailable()) - : undefined; - if ( - failedModel && - failedWindow > 0 && - contextWindow > failedWindow && - promotionTarget && - modelsAreEqual(promotionTarget, this.model) && - AIError.isContextOverflow(assistantMessage, failedWindow) - ) { - this.#removeAssistantMessageFromActiveContext(assistantMessage); - await this.#dropPersistedAssistantTurn(assistantMessage); - logger.debug("Overflow on pre-promotion model; retrying on promoted model", { - failed: `${assistantMessage.provider}/${assistantMessage.model}`, - current: `${this.model.provider}/${this.model.id}`, - }); - this.#scheduleAgentContinue({ delayMs: 100, generation }); - return COMPACTION_CHECK_CONTINUATION; - } - } - - // Case 3: Output-side incomplete — `response.incomplete` from OpenAI Responses - // (and Codex) maps to stopReason === "length". The model burned its - // `max_output_tokens` budget on reasoning/text and emitted no actionable - // deliverable. Same recovery class as overflow: promotion if available, - // otherwise compaction/handoff. Unlike overflow, the *input* is fine, so we - // allow the handoff strategy to actually run. - if (sameModel && !errorIsFromBeforeCompaction && assistantMessage.stopReason === "length") { - // Same active-context vs persisted-history split as the overflow path - // above: clear the dead turn from agent state so it cannot be replayed, - // but keep it on the branch unless promotion or compaction actually runs. - this.#removeAssistantMessageFromActiveContext(assistantMessage); - - const promoted = await this.#tryContextPromotion(assistantMessage); - if (promoted) { - await this.#dropPersistedAssistantTurn(assistantMessage); - logger.debug("Context promotion triggered by response.incomplete (length stop)", { - from: `${assistantMessage.provider}/${assistantMessage.model}`, - }); - this.#scheduleAgentContinue({ delayMs: 100, generation }); - return COMPACTION_CHECK_CONTINUATION; - } - - const incompleteCompactionSettings = this.settings.getGroup("compaction"); - if (incompleteCompactionSettings.enabled && incompleteCompactionSettings.strategy !== "off") { - logger.debug("Compaction triggered by response.incomplete (length stop, no promotion target)", { - model: `${assistantMessage.provider}/${assistantMessage.model}`, - strategy: incompleteCompactionSettings.strategy, - }); - return await this.#runRecoveryCompactionWithRollback("incomplete", assistantMessage, allowDefer, { - autoContinue, - triggerContextTokens: calculateContextTokens(assistantMessage.usage), - }); - } - // Neither promotion nor compaction is available — surface the dead-end so - // the user understands why the turn yielded with nothing. - logger.warn("response.incomplete with no recovery path (promotion + compaction both unavailable)", { - model: `${assistantMessage.provider}/${assistantMessage.model}`, - }); - return COMPACTION_CHECK_NONE; - } - - // Stale-result pass runs every turn, before any threshold gating: it is - // cheap (bails when no candidate) and independent of the compaction - // setting. - const supersedeResult = await this.#pruneStaleToolResults(); - - const compactionSettings = this.settings.getGroup("compaction"); - if (!compactionSettings.enabled || compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; - - // Case 4: Threshold - turn succeeded but context is getting large - // Skip if this was an error (non-overflow errors don't have usage data) - if (assistantMessage.stopReason === "error") return COMPACTION_CHECK_NONE; - const pruneResult = await this.#pruneToolOutputs(); - const maintenanceTokensFreed = (supersedeResult?.tokensSaved ?? 0) + (pruneResult?.tokensSaved ?? 0); - // `errorIsFromBeforeCompaction` (computed above) is the general - // "this assistant message predates the latest compaction" predicate here, - // not just an error-specific one; alias it locally so the threshold intent - // reads clearly (#3412 review). - const assistantPredatesCompaction = errorIsFromBeforeCompaction; - // An assistant that predates the latest compaction carries stale, pre-rewrite - // `usage`: the scheduled auto-continue re-enters this check with the kept - // assistant (#promptWithMessage → #checkCompaction), and its old high prompt - // count would re-trip the threshold on a freshly compacted history. Drop the - // stale provider number for those messages and let the live stored estimate - // (the floor applied below) drive the decision instead. - const assistantUsageContextTokens = assistantPredatesCompaction - ? 0 - : calculateContextTokens(assistantMessage.usage); - const storedContextTokens = this.#estimateStoredContextTokens(); - // Pruning frees bytes for the NEXT prompt; it does not change the size of - // the prompt the LLM just billed for. Earlier revisions subtracted the - // per-turn supersede/prune `tokensSaved` from the threshold input, which - // let a long-running `/goal` session sit above `compaction.thresholdTokens` - // indefinitely whenever per-turn pruning saved enough to drop the - // post-prune estimate below the user-configured trigger — the visible - // context (anchored to the same provider billing) still showed >threshold, - // but `shouldCompact` no-op'd (#3174). Anchor the initial trigger on the - // last turn's billed context tokens, floored by the post-prune - // stored-conversation estimate so a payload-compression hook still can't - // deflate the trigger. - const contextTokens = compactionContextTokens(assistantUsageContextTokens, storedContextTokens); - const postMaintenanceContextTokens = compactionContextTokens( - Math.max(0, assistantUsageContextTokens - maintenanceTokensFreed), - storedContextTokens, - ); - const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); - const shouldThresholdCompact = shouldCompact(contextTokens, contextWindow, compactionSettings); - logger.debug("Auto-compaction threshold decision", { - phase: "post-agent-end", - goalModeEnabled: this.#goalModeState?.enabled === true, - goalStatus: this.#goalModeState?.goal.status, - stopReason: assistantMessage.stopReason, - sameModel: sameModel === true, - contextWindow, - strategy: compactionSettings.strategy, - thresholdTokens, - assistantUsageContextTokens, - storedContextTokens, - resolvedContextTokens: contextTokens, - postMaintenanceContextTokens, - maintenanceTokensFreed, - shouldCompact: shouldThresholdCompact, - contextPromotionEnabled: this.settings.get("contextPromotion.enabled") === true, - }); - if (shouldThresholdCompact) { - // Try promotion first — if a larger model is available, switch instead of compacting - const promoted = await this.#tryContextPromotion(assistantMessage); - if (!promoted) { - return await this.#runAutoCompaction("threshold", false, false, allowDefer, { - autoContinue, - triggerContextTokens: postMaintenanceContextTokens, - phase: "pre_turn", - terminalTextAnswer: isTerminalTextAssistantAnswer(assistantMessage), - }); - } - logger.debug("Auto-compaction threshold satisfied but context promotion took over", { - contextTokens, - contextWindow, - model: `${assistantMessage.provider}/${assistantMessage.model}`, - }); - } - return COMPACTION_CHECK_NONE; - } #isTerminalYieldToolResult(event: { toolName: string; isError?: boolean; result?: { details?: unknown } }): boolean { if (event.toolName !== "yield" || event.isError) return false; const details = event.result?.details; @@ -12381,451 +6026,6 @@ export class AgentSession { return undefined; } - #clearPendingRecoveredRetryErrors(): void { - this.#pendingRecoveredRetryErrors = []; - } - - /** - * Durably record a terminal empty error turn (`stopReason: "error"` with no - * substantive content) that `#persistSessionMessageIfMissing` skipped, so the - * session JSONL keeps a record of why the run stopped instead of ending at the - * last tool result. A no-op for non-empty/non-error turns and idempotent via - * the already-persisted guard; the turn is dropped from active context by the - * caller (or `isProviderRefusalMessage`/`isEmptyErrorTurn` filters) so it is - * never replayed on the wire. Used by the retry-lifecycle dead-ends and the - * non-retry terminal error tail. - */ - async #persistTerminalEmptyErrorTurn(message: AssistantMessage): Promise { - await this.#waitForSessionMessagePersistence(message); - if (!isEmptyErrorTurn(message)) return; - if (this.#sessionMessageAlreadyPersisted(message)) return; - this.#appendSessionMessage(message); - } - - #retryRecoveryKind( - id: number, - switchedCredential: boolean, - switchedModel: boolean, - delayMs: number, - ): AssistantRetryRecoveryKind { - if (switchedCredential) return "credential"; - if (switchedModel) return "model"; - if (AIError.is(id, AIError.Flag.UsageLimit) && delayMs > 0) return "wait"; - return "plain"; - } - - #retryRecoveryNote(recovery: AssistantRetryRecoveryKind, rateLimited: boolean): string { - const parts: string[] = []; - if (rateLimited) { - parts.push("rate-limited"); - } else if (recovery === "plain") { - parts.push("error"); - } - if (recovery === "credential") { - parts.push("switched account"); - } else if (recovery === "model") { - parts.push("switched model"); - } else if (recovery === "wait") { - parts.push("waited"); - } - parts.push("retried"); - return parts.join("; "); - } - - async #recordPendingRecoveredRetryError( - message: AssistantMessage, - id: number, - options: { switchedCredential: boolean; switchedModel: boolean; delayMs: number }, - ): Promise { - await this.#persistTerminalEmptyErrorTurn(message); - const persistenceKey = sessionMessagePersistenceKey(message); - if (!persistenceKey) return; - let branchEntry: SessionEntry | undefined; - for (const entry of this.sessionManager.getBranch().slice().reverse()) { - if (entry.type !== "message" || entry.message.role !== "assistant") continue; - if (sessionMessagePersistenceKey(entry.message) !== persistenceKey) continue; - if (!sameMessageContent(entry.message, message) && !this.#isSameAssistantMessage(entry.message, message)) { - continue; - } - branchEntry = entry; - break; - } - if (!branchEntry) return; - if (this.#pendingRecoveredRetryErrors.some(error => error.entryId === branchEntry.id)) return; - const rateLimited = AIError.is(id, AIError.Flag.UsageLimit); - const recovery = this.#retryRecoveryKind(id, options.switchedCredential, options.switchedModel, options.delayMs); - const note = this.#retryRecoveryNote(recovery, rateLimited); - this.#pendingRecoveredRetryErrors.push({ - entryId: branchEntry.id, - persistenceKey, - recovery, - attempt: this.#retryAttempt, - note, - }); - } - - async #markPendingRecoveredRetryErrors(supersedingMessage: AssistantMessage): Promise { - if (this.#pendingRecoveredRetryErrors.length === 0) return []; - const branch = this.sessionManager.getBranch(); - const branchById = new Map(); - for (const entry of branch) { - branchById.set(entry.id, entry); - } - const recoveredAt = new Date().toISOString(); - const supersededBy: AssistantRetryRecovery["supersededBy"] = { - timestamp: supersedingMessage.timestamp, - provider: supersedingMessage.provider, - model: supersedingMessage.model, - }; - if (supersedingMessage.responseId) { - supersededBy.responseId = supersedingMessage.responseId; - } - const recoveredErrors: RecoveredRetryError[] = []; - for (const pending of this.#pendingRecoveredRetryErrors) { - let entry = branchById.get(pending.entryId); - if (entry?.type !== "message" || entry.message.role !== "assistant") { - entry = branch - .slice() - .reverse() - .find( - candidate => - candidate.type === "message" && - candidate.message.role === "assistant" && - sessionMessagePersistenceKey(candidate.message) === pending.persistenceKey, - ); - } - if (entry?.type !== "message" || entry.message.role !== "assistant") continue; - const retryRecovery: AssistantRetryRecovery = { - kind: "auto-retry", - status: "recovered", - attempt: pending.attempt, - recoveredAt, - recovery: pending.recovery, - note: pending.note, - supersededBy, - }; - entry.message.retryRecovery = retryRecovery; - recoveredErrors.push({ - entryId: entry.id, - persistenceKey: pending.persistenceKey, - note: retryRecovery.note, - retryRecovery, - }); - } - if (recoveredErrors.length > 0) { - await this.sessionManager.rewriteEntries(); - } - return recoveredErrors; - } - - async #handleEmptyAssistantStop(assistantMessage: AssistantMessage): Promise { - if (!this.#isEmptyAssistantStop(assistantMessage)) { - this.#emptyStopRetryCount = 0; - return false; - } - - if (this.#acceptTerminalEmptyStopForPrompt && assistantMessage.stopReason === "stop") { - this.#acceptTerminalEmptyStopForPrompt = false; - this.#discardAcceptedTerminalEmptyStop(assistantMessage); - this.#emptyStopRetryCount = 0; - return false; - } - - this.#emptyStopRetryCount++; - if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) { - const attempts = this.#emptyStopRetryCount - 1; - const finalError = - "Assistant returned empty stop after retry cap; try switching models or `/shake images` to remove archived frames"; - logger.warn(finalError, { - attempts, - model: assistantMessage.model, - provider: assistantMessage.provider, - }); - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt: this.#retryAttempt > 0 ? this.#retryAttempt : attempts, - finalError, - }); - this.#clearPendingRecoveredRetryErrors(); - this.#retryAttempt = 0; - this.#resolveRetry(); - // A zero-content turn carries no transcript value, while its provider usage - // can anchor the next prompt at the full failed-request size and re-trigger - // compaction at the same boundary. Remove every capped empty stop; toolUse - // orphans still need this for Anthropic message-history validity. - await this.#dropPersistedAssistantTurn(assistantMessage); - return false; - } - this.#discardAssistantTurn(assistantMessage); - this.agent.appendMessage({ - role: "developer", - content: [{ type: "text", text: this.#emptyStopRetryReminder() }], - attribution: "agent", - timestamp: Date.now(), - }); - this.#scheduleAgentContinue({ generation: this.#promptGeneration }); - return true; - } - - #isEmptyAssistantStop(assistantMessage: AssistantMessage): boolean { - switch (assistantMessage.stopReason) { - case "stop": - // Unsigned thinking alone is not actionable, but a signature is - // provider-authenticated content and makes the stop terminal. - for (const content of assistantMessage.content) { - if (content.type === "toolCall") return false; - if (content.type === "text" && hasNonWhitespace(content.text)) return false; - if (content.type === "thinking" && hasNonWhitespace(content.thinkingSignature ?? "")) return false; - } - return true; - case "toolUse": - // An orphaned toolUse stop (no tool_use block) corrupts Anthropic history: - // a later tool_result has nothing to anchor to. Thinking alone cannot anchor - // a tool_result, so it does not rescue a toolUse stop here. - for (const content of assistantMessage.content) { - if (content.type === "toolCall") return false; - if (content.type === "text" && hasNonWhitespace(content.text)) return false; - } - return true; - default: - return false; - } - } - - #emptyStopRetryReminder(): string { - return prompt.render(emptyStopRetryTemplate, { - retryCount: this.#emptyStopRetryCount, - maxRetries: EMPTY_STOP_MAX_RETRIES, - }); - } - async #handleUnexpectedAssistantStop(assistantMessage: AssistantMessage): Promise { - if (!this.settings.get("features.unexpectedStopDetection")) { - return false; - } - if (!isUnexpectedStopCandidate(assistantMessage)) { - this.#unexpectedStopRetryCount = 0; - return false; - } - - const text = assistantMessage.content - .filter((content): content is TextContent => content.type === "text") - .map(content => content.text) - .join("\n"); - if (!/\S/.test(text)) { - this.#unexpectedStopRetryCount = 0; - return false; - } - - const controller = new AbortController(); - const timeout = setTimeout(() => controller.abort(), UNEXPECTED_STOP_TIMEOUT_MS); - let classification: boolean | undefined; - try { - classification = await classifyUnexpectedStop(text, { - settings: this.settings, - registry: this.#modelRegistry, - sessionId: this.sessionId, - metadataResolver: (provider: string) => this.agent.metadataForProvider(provider), - signal: controller.signal, - }); - } finally { - clearTimeout(timeout); - } - - if (classification !== true) { - this.#unexpectedStopRetryCount = 0; - return false; - } - - this.#unexpectedStopRetryCount++; - if (this.#unexpectedStopRetryCount > UNEXPECTED_STOP_MAX_RETRIES) { - logger.warn("Assistant returned unexpected stop after retry cap", { - attempts: this.#unexpectedStopRetryCount - 1, - model: assistantMessage.model, - provider: assistantMessage.provider, - }); - this.#unexpectedStopRetryCount = 0; - return false; - } - - this.agent.appendMessage({ - role: "developer", - content: [{ type: "text", text: this.#unexpectedStopRetryReminder() }], - attribution: "agent", - timestamp: Date.now(), - }); - this.#scheduleAgentContinue({ generation: this.#promptGeneration }); - return true; - } - - #unexpectedStopRetryReminder(): string { - return prompt.render(unexpectedStopRetryTemplate, { - retryCount: this.#unexpectedStopRetryCount, - maxRetries: UNEXPECTED_STOP_MAX_RETRIES, - }); - } - - #removeAssistantMessageFromActiveContext( - assistantMessage: AssistantMessage, - reason = "assistant-context-cleanup", - ): void { - const messages = this.agent.state.messages; - const lastMessage = messages[messages.length - 1]; - const lastAssistant: AssistantMessage | undefined = lastMessage?.role === "assistant" ? lastMessage : undefined; - if (lastAssistant !== undefined && this.#isSameAssistantMessage(lastAssistant, assistantMessage)) { - this.agent.replaceMessages(messages.slice(0, -1)); - return; - } - // A miss means the failed turn is still in active context (or was never - // there); log just enough to explain why the identity check failed. - logger.debug("agent active context assistant removal missed", { - reason, - lastRole: lastMessage?.role, - candidateTimestamp: assistantMessage.timestamp, - lastTimestamp: lastAssistant?.timestamp, - candidateStopReason: assistantMessage.stopReason, - lastStopReason: lastAssistant?.stopReason, - }); - } - - /** - * Drop a recoverable assistant turn from the persisted session branch once a - * recovery path (context promotion or compaction) is committed. Waits for the - * in-flight `message_end` persistence slot first so the branch entry exists - * before we reparent past it. Active context removal is the caller's - * responsibility — recovery paths clear it eagerly so the retry never - * replays the failed turn, while no-recovery paths leave the persisted entry - * (and the user-visible transcript line) in place. - */ - async #dropPersistedAssistantTurn(assistantMessage: AssistantMessage): Promise { - await this.#waitForSessionMessagePersistence(assistantMessage); - this.#discardAssistantTurn(assistantMessage); - } - - /** - * Drop the failed assistant turn from persisted history, run - * {@link #runAutoCompaction} for an `overflow` / `incomplete` recovery, and - * restore the assistant entry if compaction did not actually commit - * anything (no usable model/preparation, hook cancel, compaction error, - * or a no-progress automatic-continuation block before any summary was - * written). - * - * Compaction has to see a clean branch — otherwise its `prepareCompaction` - * pass would keep the failed turn in the kept region and the retry would - * replay it. But a return that was not paired with a fresh compaction - * summary or a successful history rewrite means no recovery is in progress, - * even if queued user input gets drained next. Restoring the failed turn - * before that continuation preserves the visible stop reason and rebuilds the - * active assistant tail that `Agent.continue()` needs to dequeue follow-ups. - */ - async #runRecoveryCompactionWithRollback( - reason: "overflow" | "incomplete", - assistantMessage: AssistantMessage, - allowDefer: boolean, - options: { autoContinue: boolean; triggerContextTokens?: number }, - ): Promise { - const compactionEntryBefore = getLatestCompactionEntry(this.sessionManager.getBranch()); - await this.#dropPersistedAssistantTurn(assistantMessage); - const result = await this.#runAutoCompaction(reason, true, false, allowDefer, { - autoContinue: options.autoContinue, - triggerContextTokens: options.triggerContextTokens, - phase: "mid_turn", - }); - const compactionEntryAfter = getLatestCompactionEntry(this.sessionManager.getBranch()); - if (result.historyRewritten !== true && compactionEntryAfter === compactionEntryBefore) { - this.#restoreFailedAssistantTurn(assistantMessage); - } - return result; - } - - #restoreFailedAssistantTurn(assistantMessage: AssistantMessage): void { - if (!isEmptyErrorTurn(assistantMessage)) this.sessionManager.appendMessage(assistantMessage); - const lastMessage = this.agent.state.messages.at(-1); - if ( - lastMessage?.role === "assistant" && - this.#isSameAssistantMessage(lastMessage as AssistantMessage, assistantMessage) - ) { - return; - } - this.agent.appendMessage(assistantMessage); - } - - #discardAcceptedTerminalEmptyStop(assistantMessage: AssistantMessage): void { - const branch = this.sessionManager.getBranch(); - const branchEntry = branch - .slice() - .reverse() - .find( - entry => - entry.type === "message" && - entry.message.role === "assistant" && - this.#isSameAssistantMessage(entry.message, assistantMessage), - ); - const parentEntry = - branchEntry?.parentId === null || branchEntry?.parentId === undefined - ? undefined - : branch.find(entry => entry.id === branchEntry.parentId); - const prunePrompt = parentEntry?.type === "custom_message"; - - this.#removeAssistantMessageFromActiveContext(assistantMessage, "accepted-terminal-empty-stop"); - if (prunePrompt && this.agent.state.messages.at(-1)?.role === "custom") { - this.agent.replaceMessages(this.agent.state.messages.slice(0, -1)); - } - - if (!branchEntry) return; - const targetParentId = prunePrompt ? parentEntry.parentId : branchEntry.parentId; - this.#withBashBranchTransition(() => { - if (targetParentId === null) { - this.sessionManager.resetLeaf(); - } else { - this.sessionManager.branch(targetParentId); - } - }); - this.sessionManager.appendCustomEntry("accepted-terminal-empty-stop"); - } - - /** - * Drop an assistant turn from BOTH the live agent context and the persisted - * session branch (reparenting the leaf to the turn's parent), so a discarded - * turn does not resurface on reload. Used for empty/reasoning-only stops and - * the Gemini header-runaway interrupt, which must not replay a partial, - * loop-fueling thinking block. - */ - #discardAssistantTurn(assistantMessage: AssistantMessage): void { - this.#removeAssistantMessageFromActiveContext(assistantMessage); - - const branchEntry = this.sessionManager - .getBranch() - .slice() - .reverse() - .find( - entry => - entry.type === "message" && - entry.message.role === "assistant" && - this.#isSameAssistantMessage(entry.message as AssistantMessage, assistantMessage), - ); - if (!branchEntry) { - return; - } - this.#withBashBranchTransition(() => { - if (branchEntry.parentId === null) { - this.sessionManager.resetLeaf(); - } else { - this.sessionManager.branch(branchEntry.parentId); - } - }); - } - - #isSameAssistantMessage(left: AssistantMessage, right: AssistantMessage): boolean { - return ( - left === right || - (left.timestamp === right.timestamp && - left.provider === right.provider && - left.model === right.model && - left.stopReason === right.stopReason) - ); - } - #enforceRewindBeforeYield(): boolean { if (!this.#checkpointState || this.#pendingRewindReport) { return false; @@ -12870,7 +6070,7 @@ export class AgentSession { if (!checkpointState) { return; } - this.#withBashBranchTransition(() => { + this.#bash.withBranchTransition(() => { try { this.sessionManager.branchWithSummary(checkpointState.checkpointEntryId, report, { startedAt: checkpointState.startedAt, @@ -12906,8 +6106,8 @@ export class AgentSession { activeMessages.splice(0, activeMessages.length, ...sessionContext.messages); } this.agent.replaceMessages(activeMessages ?? sessionContext.messages); - this.#resetAdvisorSessionState(); - this.#syncTodoPhasesFromBranch(); + this.#advisors.resetSessionState(); + this.#todo.syncFromBranch(); this.#closeCodexProviderSessionsForHistoryRewrite(); this.#checkpointState = undefined; this.#pendingRewindReport = undefined; @@ -12949,7 +6149,7 @@ export class AgentSession { logger.debug("Plan mode convergence: reminder cap reached; yielding to user"); return false; } - const hasRequiredTools = this.#toolRegistry.has("ask") && this.#toolRegistry.has("write"); + const hasRequiredTools = this.#tools.registry.has("ask") && this.#tools.registry.has("write"); if (!hasRequiredTools) { logger.warn("Plan mode enforcement skipped because ask/write tools are unavailable", { activeToolNames: this.agent.state.tools.map(tool => tool.name), @@ -12981,397 +6181,6 @@ export class AgentSession { return true; } - /** - * Render context shared by the eager todo/task preludes. `toolRefs` resolves each - * tool's wire name (matching `buildSystemPrompt`'s `toolRefs`) so the reminder names - * the tool the model actually sees when an extension renames it; `taskBatch` gates - * batch-call guidance that would steer toward a failing call shape when `task.batch` - * is off (the flat single-spawn schema rejects `tasks`/`context`). - */ - #buildEagerPreludeContext(): { toolRefs: Record; taskBatch: boolean } { - const wireName = (name: string): string => { - const tool = this.#toolRegistry.get(name); - return typeof tool?.customWireName === "string" ? tool.customWireName : name; - }; - return { - toolRefs: { task: wireName("task"), todo: wireName("todo") }, - taskBatch: this.settings.get("task.batch"), - }; - } - - #createEagerTodoPrelude( - promptText: string | undefined, - ): { message: AgentMessage; toolChoice?: ToolChoice } | undefined { - const mode = this.settings.get("todo.eager"); - const todosEnabled = this.settings.get("todo.enabled"); - if (mode === "default" || !todosEnabled) { - return undefined; - } - - if (this.#planModeState?.enabled) { - return undefined; - } - if (this.getTodoPhases().length > 0) { - return undefined; - } - - // Only inject on the first user message of the conversation. Subsequent user - // turns must not receive the eager todo reminder — they often correct, clarify, - // or redirect the prior task, and forcing a brand-new todo list there is wrong. - // When `promptText` is undefined (post-compaction re-injection) there is no fresh - // user message to gate on, so skip the first-message and prompt-suffix checks. - if (promptText !== undefined) { - const hasPriorUserMessage = this.agent.state.messages.some(m => m.role === "user"); - if (hasPriorUserMessage) { - return undefined; - } - - const trimmedPromptText = promptText.trimEnd(); - if (trimmedPromptText.endsWith("?") || trimmedPromptText.endsWith("!")) { - return undefined; - } - } - - // Must check the active tool set, not just the registry: a registered - // tool can be hidden from the exposed tools (e.g. unmounted under the - // xd:// transport). Forcing a named tool_choice for an inactive tool makes - // the provider reject the request (HTTP 400). - if (!this.getActiveToolNames().includes("todo")) { - logger.warn("Eager todo enforcement skipped because todo is not active", { - activeToolNames: this.getActiveToolNames(), - }); - return undefined; - } - - const message: AgentMessage = { - role: "custom", - customType: "eager-todo-prelude", - content: prompt.render(eagerTodoPrompt, { ...this.#buildEagerPreludeContext(), forced: mode === "always" }), - display: false, - attribution: "agent", - timestamp: Date.now(), - }; - // `preferred` suggests a todo list (reminder only); `always` also forces the - // `todo` tool on the first turn — the previous boolean-on behavior. Post-compaction - // re-injection (`promptText === undefined`) is always reminder-only: forcing a tool - // onto the auto-resumed turn would override the agent's in-flight action. - if (promptText === undefined || mode === "preferred") { - return { message }; - } - const todoToolChoice = buildNamedToolChoice("todo", this.model); - if (!todoToolChoice) { - // `always` on a model that can't be forced degrades to reminder-only (no - // tool_choice). For `todo.eager: true` users migrated to `always`, such - // models now receive the first-turn reminder where they previously got - // nothing (see the CHANGELOG entry); `always ⊇ preferred` is preserved. - logger.warn( - "Eager todo proceeding with the reminder only because the current model does not support a forced todo tool_choice", - { modelApi: this.model?.api, modelId: this.model?.id }, - ); - return { message }; - } - return { message, toolChoice: todoToolChoice }; - } - - #createEagerTaskPrelude(promptText: string | undefined): AgentMessage | undefined { - if (this.settings.get("task.eager") !== "always") return undefined; - // Main agent only: subagents keep `task` active (the parent only filters `todo`), - // so a salient delegate-reminder there would amplify nested fan-out. Gate on the - // resolved agent kind, not the id, so a top-level session with a custom `agentId` - // still gets the reminder. - if (this.#agentKind === "sub") return undefined; - if (this.#planModeState?.enabled) return undefined; - // First-message-only gates are skipped post-compaction (`promptText === undefined`), - // where there is no fresh user message to suppress the reminder for. - if (promptText !== undefined) { - if (this.agent.state.messages.some(m => m.role === "user")) return undefined; - const trimmed = promptText.trimEnd(); - if (trimmed.endsWith("?") || trimmed.endsWith("!")) return undefined; - } - if (!this.getActiveToolNames().includes("task")) return undefined; - return { - role: "custom", - customType: "eager-task-prelude", - content: prompt.render(eagerTaskPrompt, this.#buildEagerPreludeContext()), - display: false, - attribution: "agent", - timestamp: Date.now(), - }; - } - - /** - * Build the eager task/todo reminders to re-inject on the auto-continuation turn that - * follows a compaction. The first-message preludes are the oldest messages, so - * compaction summarizes them away and the agent silently loses the delegate-via-tasks - * and phased-todo guidance mid-work; this re-asserts them, reminder-only (the todo - * builder drops its forced tool_choice when `promptText` is undefined). Each builder - * still applies its own mode / agent-kind / plan-mode / tool-active / surviving-todo - * gates, so an empty array means nothing currently warrants a nudge. - */ - #buildPostCompactionEagerNudges(): AgentMessage[] { - const nudges: AgentMessage[] = []; - const todo = this.#createEagerTodoPrelude(undefined); - if (todo) nudges.push(todo.message); - const task = this.#createEagerTaskPrelude(undefined); - if (task) nudges.push(task); - return nudges; - } - /** - * Check if agent stopped with incomplete todos and prompt to continue. - */ - async #checkTodoCompletion(message: AssistantMessage): Promise { - // Skip todo reminders when the most recent turn was driven by an explicit user force — - // the user wanted exactly that tool, not a follow-up nag about incomplete todos. - const lastServedLabel = this.#toolChoiceQueue.consumeLastServedLabel(); - if (lastServedLabel === "user-force") { - return false; - } - - // Plan mode owns convergence via #enforcePlanModeDecisionAtSettle (remind → - // cap → yield). Todo reminders must not re-wake a turn the cap intends to - // yield to the user. The label is already consumed above, so no leak. - if (this.#planModeState?.enabled) { - return false; - } - - // Suppress within a self-continuation chain: if the agent's last turn was driven by a - // prior reminder (and the agent took no tool-level action since), do not re-ping. - // The agent has already acknowledged; further escalation just wastes context and - // pressures the agent into busy-work or destructive ops (issue #2590). - if (this.#todoReminderAwaitingProgress) { - logger.debug("Todo completion: prior reminder still awaiting agent action; staying silent", { - attempt: this.#todoReminderCount, - }); - return false; - } - - const remindersEnabled = this.settings.get("todo.reminders"); - const todosEnabled = this.settings.get("todo.enabled"); - if (!remindersEnabled || !todosEnabled) { - this.#todoReminderCount = 0; - this.#todoReminderAwaitingProgress = false; - return false; - } - - const remindersMax = this.settings.get("todo.remindersMax"); - if (this.#todoReminderCount >= remindersMax) { - logger.debug("Todo completion: max reminders reached", { count: this.#todoReminderCount }); - return false; - } - - const phases = this.getTodoPhases(); - if (phases.length === 0) { - this.#todoReminderCount = 0; - this.#todoReminderAwaitingProgress = false; - return false; - } - - const incompleteByPhase = phases - .map(phase => ({ - name: phase.name, - tasks: phase.tasks - .filter( - (task): task is TodoItem & { status: "pending" | "in_progress" } => - task.status === "pending" || task.status === "in_progress", - ) - .map(task => ({ content: task.content, status: task.status })), - })) - .filter(phase => phase.tasks.length > 0); - const incomplete = incompleteByPhase.flatMap(phase => phase.tasks); - if (incomplete.length === 0) { - this.#todoReminderCount = 0; - this.#todoReminderAwaitingProgress = false; - return false; - } - - if (isAwaitingUserAnswer(message)) { - logger.debug("Todo completion: assistant is waiting for user input; skipping reminder", { - incomplete: incomplete.length, - }); - return false; - } - - // Background async jobs (bash/task) owned by this agent re-wake the loop - // when they complete: the result delivery enqueues an async-result - // follow-up that continues the run, and todos are re-evaluated at that - // settle. A stop with such a job in flight is a scheduling pause, not - // abandonment — stay silent instead of nagging. - if (this.#hasPendingAsyncWake()) { - logger.debug("Todo completion: async jobs in flight will re-wake the loop; skipping reminder", { - incomplete: incomplete.length, - }); - return false; - } - - // Build reminder message - this.#todoReminderCount++; - const todoList = incompleteByPhase - .map(phase => `- ${phase.name}\n${phase.tasks.map(task => ` - ${task.content}`).join("\n")}`) - .join("\n"); - const reminder = - `\n` + - `You stopped with ${incomplete.length} incomplete todo item(s):\n${todoList}\n\n` + - `Please continue working on these tasks or mark them complete if finished.\n` + - `(Reminder ${this.#todoReminderCount}/${remindersMax})\n` + - ``; - - logger.debug("Todo completion: sending reminder", { - incomplete: incomplete.length, - attempt: this.#todoReminderCount, - }); - - // Emit event for UI to render notification - await this.#emitSessionEvent({ - type: "todo_reminder", - todos: incomplete, - attempt: this.#todoReminderCount, - maxAttempts: remindersMax, - }); - - const reminderMessage: Message = { - role: "developer", - content: [{ type: "text", text: reminder }], - attribution: "agent", - timestamp: Date.now(), - }; - - // A stop-time reminder starts a fresh reminder runway. Without resetting - // the mid-run counter here, a run that stopped just below the threshold - // would spend its stale pre-reminder count and fire "Mid-run reminder 2/3" - // after only a little post-reminder work. - this.#mutationsSinceLastTodoTouch = 0; - this.#todoReminderAwaitingProgress = true; - // Inject reminder and persist it so the JSONL transcript matches model context. - this.agent.appendMessage(reminderMessage); - this.sessionManager.appendMessage(reminderMessage); - this.#scheduleAgentContinue({ generation: this.#promptGeneration }); - return true; - } - - /** - * Build the next mid-run todo reconciliation nudge when the agent has landed - * {@link MID_RUN_TODO_NUDGE_MUTATION_THRESHOLD} mutating tool results without - * invoking the `todo` tool and incomplete items remain. Returns the hidden - * (`display: false`) custom message when it should fire, or `null` to skip. - * Called once per turn via the aside provider; mutates internal counters when - * it fires so the caller does not need to track delivery state. - * - * Deliberately a SEPARATE concept from {@link #checkTodoCompletion}'s - * stop-time reminder: this is a gentle model-only hint (no `todo_reminder` - * event, no TUI render, no escalation counter, own per-cycle budget), while - * the stop-time reminder is the user-visible escalation ladder. Without this - * nudge, long runs drive the live HUD to `0/N` until the final stop, then - * batch-flip to `N/N` (issue #3651). - */ - #takeMidRunTodoNudge(): AgentMessage | null { - if (this.#mutationsSinceLastTodoTouch < MID_RUN_TODO_NUDGE_MUTATION_THRESHOLD) return null; - if (this.#midRunNudgeCount >= MID_RUN_TODO_NUDGE_MAX_PER_CYCLE) return null; - if (!this.settings.get("todo.enabled")) return null; - if (!this.settings.get("todo.reminders")) return null; - // Plan-mode runs are authoring a plan file, not implementing it; todos - // don't apply, mirroring {@link #createEagerTodoPrelude}. - if (this.#planModeState?.enabled) return null; - // Tool discovery / explicit active-tool lists can hide `todo` from this - // run while `todo.enabled` remains true (e.g. `setActiveToolsByName` - // restricting the slate). Mirror {@link #createEagerTodoPrelude}'s - // guard so we never ask the model to call a tool that is not in its - // schema — the request would fabricate an unknown tool call. - if (!this.getActiveToolNames().includes("todo")) return null; - - const incomplete = this.getTodoPhases() - .flatMap(phase => phase.tasks) - .filter(task => task.status === "pending" || task.status === "in_progress"); - if (incomplete.length === 0) return null; - - // Reset the mutation counter so the nudge has another full runway before - // the next fire; #midRunNudgeCount caps total nudges per prompt cycle. - this.#mutationsSinceLastTodoTouch = 0; - this.#midRunNudgeCount++; - - const { toolRefs } = this.#buildEagerPreludeContext(); - const reminder = prompt.render(midRunTodoNudgePrompt, { - toolRefs, - incompleteCount: incomplete.length, - plural: incomplete.length !== 1, - }); - - logger.debug("Mid-run todo nudge fired", { - incomplete: incomplete.length, - nudge: this.#midRunNudgeCount, - }); - - return { - role: "custom", - customType: MID_RUN_TODO_NUDGE_MESSAGE_TYPE, - content: reminder, - display: false, - attribution: "agent", - timestamp: Date.now(), - }; - } - - /** - * Attempt context promotion to a larger model. - * Returns true if promotion succeeded (caller should retry without compacting). - */ - async #tryContextPromotion(assistantMessage: AssistantMessage): Promise { - const currentModel = this.model; - if (!currentModel) return false; - // The overflow/length error may have come from a model the user already - // switched away from; only promote when the failing turn was this model. - if (assistantMessage.provider !== currentModel.provider || assistantMessage.model !== currentModel.id) - return false; - return this.#promoteContextModel(); - } - - /** - * Switch to a larger-context sibling when context promotion is enabled and a - * target with a strictly larger window (and a usable key) exists. Returns true - * when the model was switched, so the caller can retry without compacting. - * Message-independent core shared by the post-turn overflow path - * ({@link #tryContextPromotion}) and the pre-prompt threshold path - * ({@link #runPrePromptCompactionIfNeeded}). - */ - async #promoteContextModel(): Promise { - const promotionSettings = this.settings.getGroup("contextPromotion"); - if (!promotionSettings.enabled) return false; - const currentModel = this.model; - if (!currentModel) return false; - const contextWindow = currentModel.contextWindow ?? 0; - if (contextWindow <= 0) return false; - const targetModel = await this.#resolveContextPromotionTarget(currentModel, contextWindow); - if (!targetModel) return false; - - try { - await this.setModelTemporary(targetModel, undefined, { ephemeral: true }); - logger.debug("Context promotion switched model on overflow", { - from: `${currentModel.provider}/${currentModel.id}`, - to: `${targetModel.provider}/${targetModel.id}`, - }); - return true; - } catch (error) { - logger.warn("Context promotion failed", { - from: `${currentModel.provider}/${currentModel.id}`, - to: `${targetModel.provider}/${targetModel.id}`, - error: String(error), - }); - return false; - } - } - - async #resolveContextPromotionTarget(currentModel: Model, contextWindow: number): Promise { - const availableModels = this.#modelRegistry.getAvailable(); - if (availableModels.length === 0) return undefined; - - const candidate = this.#resolveContextPromotionConfiguredTarget(currentModel, availableModels); - if (!candidate) return undefined; - if (modelsAreEqual(candidate, currentModel)) return undefined; - if (candidate.contextWindow == null || candidate.contextWindow <= contextWindow) return undefined; - const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId); - if (!apiKey) return undefined; - return candidate; - } - #setModelWithProviderSessionReset(model: Model): void { const currentModel = this.model; if (currentModel) { @@ -13498,3111 +6307,39 @@ export class AgentSession { } } - #normalizeProviderReplayValue(value: unknown): unknown { - if (Array.isArray(value)) { - return value.map(item => this.#normalizeProviderReplayValue(item)); - } - if (value && typeof value === "object") { - return Object.fromEntries( - Object.entries(value).map(([key, entryValue]) => [key, this.#normalizeProviderReplayValue(entryValue)]), - ); - } - return value; - } - - #normalizeSessionMessageForProviderReplay(message: AgentMessage): unknown { - switch (message.role) { - case "user": - case "developer": - return { - role: message.role, - content: this.#normalizeProviderReplayValue(message.content), - providerPayload: message.providerPayload, - }; - case "assistant": { - const isResponsesFamilyMessage = - message.api === "openai-responses" || message.api === "openai-codex-responses"; - return { - role: message.role, - content: - isResponsesFamilyMessage && Array.isArray(message.content) - ? message.content.flatMap(block => { - if (block.type === "thinking") { - return []; - } - if (block.type === "toolCall") { - return [ - { - type: block.type, - id: block.id, - name: block.name, - arguments: block.arguments, - }, - ]; - } - if (block.type === "text") { - return [{ type: block.type, text: block.text, textSignature: block.textSignature }]; - } - return [this.#normalizeProviderReplayValue(block)]; - }) - : this.#normalizeProviderReplayValue(message.content), - api: message.api, - provider: message.provider, - model: message.model, - stopReason: message.stopReason, - errorMessage: message.errorMessage, - providerPayload: isResponsesFamilyMessage ? undefined : message.providerPayload, - }; - } - case "toolResult": - return { - role: message.role, - toolName: message.toolName, - toolCallId: message.toolCallId, - isError: message.isError, - content: this.#normalizeProviderReplayValue(message.content), - }; - case "bashExecution": - return { - role: message.role, - command: message.command, - output: message.output, - exitCode: message.exitCode, - cancelled: message.cancelled, - meta: message.meta - ? { - truncation: this.#normalizeProviderReplayValue(message.meta.truncation), - limits: this.#normalizeProviderReplayValue(message.meta.limits), - diagnostics: message.meta.diagnostics - ? this.#normalizeProviderReplayValue({ - summary: message.meta.diagnostics.summary, - messages: message.meta.diagnostics.messages, - }) - : undefined, - } - : undefined, - excludeFromContext: message.excludeFromContext, - }; - case "pythonExecution": - return { - role: message.role, - code: message.code, - output: message.output, - exitCode: message.exitCode, - cancelled: message.cancelled, - meta: message.meta - ? { - truncation: this.#normalizeProviderReplayValue(message.meta.truncation), - limits: this.#normalizeProviderReplayValue(message.meta.limits), - diagnostics: message.meta.diagnostics - ? this.#normalizeProviderReplayValue({ - summary: message.meta.diagnostics.summary, - messages: message.meta.diagnostics.messages, - }) - : undefined, - } - : undefined, - excludeFromContext: message.excludeFromContext, - }; - case "custom": - case "hookMessage": - return { - role: message.role, - customType: message.customType, - content: this.#normalizeProviderReplayValue(message.content), - }; - case "branchSummary": - return { role: message.role, summary: message.summary }; - case "compactionSummary": - return { - role: message.role, - summary: message.summary, - providerPayload: message.providerPayload, - }; - case "fileMention": - return { - role: message.role, - files: message.files.map(file => ({ - path: file.path, - content: file.content, - image: file.image, - })), - }; - default: - return this.#normalizeProviderReplayValue(message); - } - } - - #didSessionMessagesChange(previousMessages: AgentMessage[], nextMessages: AgentMessage[]): boolean { - if (previousMessages.length !== nextMessages.length) return true; - return previousMessages.some( - (message, i) => - !Bun.deepEquals( - this.#normalizeSessionMessageForProviderReplay(message), - this.#normalizeSessionMessageForProviderReplay(nextMessages[i]), - ), - ); - } - - #getModelKey(model: Model): string { - return `${model.provider}/${model.id}`; - } - - #formatRoleModelValue( - role: string, - model: Model, - selectorOverride?: string, - thinkingLevelOverride?: ThinkingLevel, - ): string { - const modelKey = selectorOverride ?? `${model.provider}/${model.id}`; - if (thinkingLevelOverride !== undefined) { - return formatModelSelectorValue(modelKey, thinkingLevelOverride); - } - const existingRoleValue = this.settings.getModelRole(role); - if (!existingRoleValue) return modelKey; - - const thinkingLevel = extractExplicitThinkingSelector(existingRoleValue, this.settings, { - isLiteralModelId: (provider, id) => this.#modelRegistry.find(provider, id) !== undefined, - }); - return formatModelSelectorValue(modelKey, thinkingLevel); - } - #resolveConfiguredModelTarget( - configuredTarget: string | undefined, - currentModel: Model, - availableModels: Model[], - ): Model | undefined { - const trimmedTarget = configuredTarget?.trim(); - if (!trimmedTarget) return undefined; - - const parsed = parseModelString(trimmedTarget, { - allowMaxSuffix: true, - allowAutoAlias: true, - isLiteralModelId: (provider, id) => - availableModels.some(model => model.provider === provider && model.id === id), - }); - if (parsed) { - const explicitModel = availableModels.find(m => m.provider === parsed.provider && m.id === parsed.id); - if (explicitModel) return explicitModel; - } - - return availableModels.find(m => m.provider === currentModel.provider && m.id === trimmedTarget); - } - - #resolveContextPromotionConfiguredTarget(currentModel: Model, availableModels: Model[]): Model | undefined { - return this.#resolveConfiguredModelTarget(currentModel.contextPromotionTarget, currentModel, availableModels); - } - - #resolveCompactionConfiguredTarget(currentModel: Model, availableModels: Model[]): Model | undefined { - return this.#resolveConfiguredModelTarget(currentModel.compactionModel, currentModel, availableModels); - } - - #resolveRoleModelFull( - role: string, - availableModels: Model[], - currentModel: Model | undefined, - ): ResolvedModelRoleValue { - const roleModelStr = - role === "default" - ? (this.settings.getModelRole("default") ?? - (currentModel ? `${currentModel.provider}/${currentModel.id}` : undefined)) - : this.settings.getModelRole(role); - - if (!roleModelStr) { - return { model: undefined, thinkingLevel: undefined, explicitThinkingLevel: false, warning: undefined }; - } - - return resolveModelRoleValue(roleModelStr, availableModels, { - settings: this.settings, - matchPreferences: getModelMatchPreferences(this.settings), - }); - } - - #getCompactionModelCandidates(availableModels: Model[], filter?: (model: Model) => boolean): Model[] { - return this.#resolveCompactionModelCandidates(this.model, availableModels, filter); - } - - #resolveCompactionModelCandidates( - preferredModel: Model | null | undefined, - availableModels: Model[], - filter?: (model: Model) => boolean, - ): Model[] { - const candidates: Model[] = []; - const seen = new Set(); - - const addCandidate = (model: Model | undefined): void => { - if (!model) return; - const key = this.#getModelKey(model); - if (seen.has(key)) return; - seen.add(key); - // `seen` still tracks rejected models so the largest-context fallback - // scan below doesn't reintroduce them; the filter just suppresses - // inclusion in this caller's candidate chain. - if (filter && !filter(model)) return; - candidates.push(model); - }; - - if (preferredModel) { - addCandidate(this.#resolveCompactionConfiguredTarget(preferredModel, availableModels)); - } - addCandidate(preferredModel ?? undefined); - for (const role of MODEL_ROLE_IDS) { - addCandidate(this.#resolveRoleModelFull(role, availableModels, preferredModel ?? undefined).model); - } - - const sortedByContext = [...availableModels].sort((a, b) => (b.contextWindow ?? 0) - (a.contextWindow ?? 0)); - for (const model of sortedByContext) { - if (!seen.has(this.#getModelKey(model))) { - addCandidate(model); - break; - } - } - - return candidates; - } - - #buildCompactionAuthError(): Error { - const currentModel = this.model; - if (!currentModel) { - return new Error( - "Compaction requires a model with usable credentials, but no authenticated compaction model is available.", - ); - } - return new Error( - `Compaction requires usable credentials for ${currentModel.provider}/${currentModel.id}. ` + - `Configure ${currentModel.provider} credentials or assign an authenticated fallback role such as modelRoles.smol.`, - ); - } - - async #compactWithFallbackModel( - preparation: CompactionPreparation, - customInstructions: string | undefined, - signal: AbortSignal, - options?: SummaryOptions, - precomputedCandidates?: Model[], - ): Promise { - const candidates = - precomputedCandidates ?? this.#getCompactionModelCandidates(this.#modelRegistry.getAvailable()); - const telemetry = resolveTelemetry(this.agent.telemetry, this.sessionId); - - for (const candidate of candidates) { - const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId); - if (!apiKey) continue; - - try { - return await compact( - this.#obfuscatePreparationForProvider(preparation), - candidate, - this.#modelRegistry.resolver(candidate, this.sessionId), - this.#obfuscateTextForProvider(customInstructions), - signal, - { - ...options, - metadata: this.agent.metadataForProvider(candidate.provider), - convertToLlm: messages => this.#convertToLlmForSideRequest(messages), - telemetry, - // Honor the user's /model thinking selection (incl. `off`) on - // the manual `/compact` path. Clamped per-model inside compact() - // via resolveCompactionEffort so unsupported-effort models - // (xai-oauth/grok-build) don't trip requireSupportedEffort. - thinkingLevel: this.thinkingLevel, - tools: this.agent.state.tools, - sessionId: this.sessionId, - promptCacheKey: this.sessionId, - providerSessionState: this.#providerSessionState, - // Route every summarization HTTP request through the - // session's side-stream transport so the provider - // concurrency cap (e.g. providers.ollama-cloud.maxConcurrency) - // brackets compaction the same way it brackets the live - // agent turn — without this, multiple ollama-cloud - // subagents auto/manually compacting issued uncapped - // summary requests in parallel (chatgpt-codex review on - // #3751). - completeImpl: async (requestModel, requestContext, requestOptions) => { - const stream = await this.#sideStreamFn(requestModel, requestContext, requestOptions); - return stream.result(); - }, - }, - ); - } catch (error) { - if (!AIError.is(AIError.classify(error, candidate.api), AIError.Flag.AuthFailed)) { - throw error; - } - } - } - - throw this.#buildCompactionAuthError(); - } - - async #prepareCompactionFromHooks( - preparation: CompactionPreparation, - hookCompaction: CompactionResult | undefined, - ): Promise< - | { - kind: "fromHook"; - summary: string; - shortSummary: string | undefined; - firstKeptEntryId: string; - tokensBefore: number; - details: unknown; - preserveData: Record | undefined; - } - | { - kind: "needsLlm"; - hookContext: string[] | undefined; - hookPrompt: string | undefined; - preserveData: Record | undefined; - } - > { - let hookContext: string[] | undefined; - let hookPrompt: string | undefined; - let preserveData: Record | undefined; - - if (!hookCompaction && this.#extensionRunner?.hasHandlers("session.compacting")) { - const compactMessages = preparation.messagesToSummarize.concat(preparation.turnPrefixMessages); - const result = (await this.#extensionRunner.emit({ - type: "session.compacting", - sessionId: this.sessionId, - messages: compactMessages, - })) as { context?: string[]; prompt?: string; preserveData?: Record } | undefined; - - hookContext = result?.context; - hookPrompt = result?.prompt; - preserveData = result?.preserveData; - } - - const memoryBackendContext = await this.#collectMemoryBackendContext(preparation); - if (memoryBackendContext) { - hookContext = hookContext ? [...hookContext, memoryBackendContext] : [memoryBackendContext]; - } - - if (hookCompaction) { - preserveData ??= hookCompaction.preserveData; - return { - kind: "fromHook", - summary: hookCompaction.summary, - shortSummary: hookCompaction.shortSummary, - firstKeptEntryId: hookCompaction.firstKeptEntryId, - tokensBefore: hookCompaction.tokensBefore, - details: hookCompaction.details, - preserveData, - }; - } - - return { kind: "needsLlm", hookContext, hookPrompt, preserveData }; - } - - /** - * Cap on snapcompact frames the post-compaction context can carry without - * busting the model window. Mirrors the per-frame token charge used by the - * projection ({@link snapcompact.FRAME_TOKEN_ESTIMATE}, the conservative - * high-res Anthropic ceiling), so picking `maxFrames` from this helper makes - * {@link #projectSnapcompactContextTokens} succeed by construction. - * - * Skip vs. cap use different reserves on purpose. The **skip** decision - * (return `0`) trips only when kept-recent plus non-message tokens already - * eat the entire `ctxWindow − reserve` envelope: at that point no archive - * shape — frame-bearing or text-only — can fit, and the caller MUST - * shortcut to the LLM summarizer instead of re-running snapcompact to - * re-emit the "could not bring the context under the limit" warning every - * threshold tick. The **cap** calculation subtracts a shape-aware reserve - * (`2 × geometry(shape).capacity` chars worth of text edges, billed at the - * tiktoken cl100k baseline, plus a 2k summary-template allowance) sized - * from the same `shape` snapcompact will use, so the projection still - * passes once frames land — but it MUST NOT gate the skip decision, since - * a frame-less archive (`text.length <= 2 * edgeCap` short-circuit in - * `planArchive`) typically costs only a few hundred tokens of summary - * lead and would fit under residual headroom far smaller than the cap - * reserve (chatgpt-codex reviews on #3249). - * - * Returns `1` when the frame charge would overflow but the text-only path - * still has room: snapcompact's planner picks the frame-less layout - * automatically when the discarded text fits in the edges, so giving it - * the minimum cap lets it succeed instead of being skipped outright. - * - * Without this cap, the bundled `MAX_FRAMES_DEFAULT = 80` × 5024 tokens = - * ~402k frame-token projection always overflows any sub-1M-token window - * (issue #3247). - */ - #computeSnapcompactMaxFrames(preparation: CompactionPreparation, settings: CompactionSettings): number { - const ctxWindow = this.model?.contextWindow ?? 0; - if (ctxWindow <= 0) return Math.min(snapcompact.MAX_FRAMES_DEFAULT, snapcompact.maxFramesForDataBudget()); - const reserve = effectiveReserveTokens(ctxWindow, settings); - let baseTokens = computeNonMessageTokens(this); - for (const message of preparation.recentMessages) { - baseTokens += estimateTokens(message); - } - const totalBudget = ctxWindow - reserve; - // Skip iff there is no headroom whatsoever; a text-only archive costs - // far less than the cap reserve below, so any positive residual is - // worth attempting and the projection guard catches actual overflow. - if (baseTokens >= totalBudget) return 0; - // Cap reserve mirrors what `estimateTokens(summaryMessage)` will charge - // when frames > 0: `countTokens(summaryTemplate ‖ textHead ‖ textTail)` - // plus `numFrames × FRAME_TOKEN_ESTIMATE`. Resolve the shape this - // snapcompact pass will actually use (matches the `shape` argument - // passed to `snapcompact.compact` in the auto and manual paths) so the - // text-edge cost reflects the live frame geometry rather than a fixed - // approximation. Reviewer (chatgpt-codex on #3249): a 4k reserve - // undersized the ~7k text-edge cost on the default Anthropic - // 11on16-bw shape, so the projection then rejected the `maxFrames` - // the cap had picked and the warning loop reappeared. - // - // - `textHead` and `textTail` each consume up to `geometry.capacity` - // chars when frames > 0 (one HQ-capacity page per edge: see - // `TEXT_EDGE_PAGES = 1` in `planArchive`), so 2 × capacity chars - // total. Per-shape capacity: Anthropic 11on16-bw ~13.9k, Opus - // 1932px ~21k, Gemini 8on22-bw 2048px ~23.8k, OpenAI 1568px ~13.9k. - // - tiktoken cl100k ≈ 4 chars/token on ASCII (verified empirically - // for prose, code, and JSON); a 1.15 multiplier absorbs tokenizer - // drift on denser content (e.g. dense JSON / tool-result blobs). - // - Summary template (intro + FILES section + grid notes) bills - // ~2k tokens for typical sessions. - const shape = snapcompact.resolveShape(this.model, this.settings.get("snapcompact.shape")); - const edgeCap = snapcompact.geometry(shape).capacity; - const textEdgeTokens = Math.ceil((2 * edgeCap * 1.15) / 4); - const SUMMARY_TEMPLATE_TOKENS = 2000; - const capReserve = textEdgeTokens + SUMMARY_TEMPLATE_TOKENS; - const frameBudget = totalBudget - baseTokens - capReserve; - if (frameBudget < snapcompact.FRAME_TOKEN_ESTIMATE) return 1; - return Math.min( - Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), - snapcompact.MAX_FRAMES_DEFAULT, - snapcompact.maxFramesForDataBudget(), - ); - } - - #snapcompactFramePayloadBytes(result: snapcompact.CompactionResult): number { - const archive = snapcompact.getPreservedArchive(result.preserveData); - return archive ? snapcompact.frameDataBytes(archive.frames) : 0; - } - - /** - * Project the post-compaction context size of a snapcompact result: kept - * recent messages + the summary message with its re-attached frames + the - * fixed non-message overhead (system prompt + tools). Mirrors how the - * compacted context is rebuilt, so the estimate matches the wire shape, and - * lets the caller decide whether snapcompact brought the context under the - * window or should fall back to an LLM summary. - */ - #projectSnapcompactContextTokens(preparation: CompactionPreparation, result: snapcompact.CompactionResult): number { - const archive = snapcompact.getPreservedArchive(result.preserveData); - const blocks = archive - ? snapcompact.historyBlocks(archive, { maxFrameDataBytes: snapcompact.FRAME_DATA_BYTES_BUDGET }) - : undefined; - const summaryMessage = createCompactionSummaryMessage( - result.summary, - result.tokensBefore, - new Date().toISOString(), - result.shortSummary, - undefined, - undefined, - blocks, - ); - let tokens = computeNonMessageTokens(this) + estimateTokens(summaryMessage); - for (const message of preparation.recentMessages) { - tokens += estimateTokens(message); - } - return tokens; - } - - /** - * Post-maintenance progress check for the context-full / snapcompact tail. - * - * After `appendCompaction` rewrote history and `replaceMessages` swapped in the - * compacted context, measure the residual context off the live message set and - * decide whether maintenance actually created headroom. Mirrors the shake - * recovery-band logic (#2275): a session whose single most-recent turn already - * blows the threshold cannot be reduced by compaction (findCutPoint keeps that - * turn verbatim), so re-firing on the next agent_end just thrashes. We only - * report progress when residual context lands at or below - * `COMPACTION_RECOVERY_BAND × threshold` — a band that sits strictly under the - * compaction threshold, so reaching it guarantees the next turn cannot - * re-trip threshold compaction. - * - * When the model/window is unknown we cannot evaluate the band, so we - * optimistically allow the continuation (preserving prior behavior). - */ - #compactionCreatedHeadroom(): boolean { - const contextWindow = this.model?.contextWindow ?? 0; - if (contextWindow <= 0) return true; - const compactionSettings = this.settings.getGroup("compaction"); - const residualTokens = compactionContextTokens( - this.getContextUsage({ contextWindow })?.tokens ?? 0, - this.#estimateStoredContextTokens(), - ); - const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); - const recoveryBand = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); - // Residual at/below the band is authoritative headroom: the band sits - // strictly under the compaction threshold, so the next turn cannot - // re-trip threshold compaction regardless of how little this pass shaved. - // Don't add a secondary "smaller than the trigger" guard — when stale/ - // tool-output pruning already dropped context under the band before this - // pass, the trigger is itself sub-band, and requiring a strict reduction - // would suppress a valid continuation and emit a false no-progress warning - // even though compaction left the session safe. - return residualTokens <= recoveryBand; - } - - /** - * Retry-side counterpart to {@link #compactionCreatedHeadroom}. An - * overflow/incomplete recovery only needs the rebuilt prompt to *fit* the - * window again — it does not have to land under the compaction threshold, let - * alone the stricter `COMPACTION_RECOVERY_BAND × threshold` hysteresis the - * auto-continue thrash guard uses. Reusing the band here turned recoverable - * overflows into manual dead-ends: a 200k-window prompt compacted from - * overflow down to ~150k is comfortably retryable, but sits above - * `0.8 × 170k = 136k` and was wrongly refused (PR #3412 review). - * - * Measures residual context against the usable budget (`contextWindow - reserve`). - * The default absolute reserve can exceed bundled small-context windows, or - * nearly consume a 16k-class window; those known-impossible defaults fall - * back to the proportional 15% reserve. Explicit valid reserves still define - * the usable prompt budget so retries do not enter headroom the user - * intentionally reserved. Callers MUST - * invoke this AFTER dropping the failed assistant from `this.messages`, so - * the just-failed turn (which the retry prompt will not include) is excluded - * from the estimate. - * - * When the model/window is unknown we cannot evaluate the budget, so we - * optimistically allow the retry (preserving prior behavior). - */ - #compactionCreatedRetryFit(): boolean { - const contextWindow = this.model?.contextWindow ?? 0; - if (contextWindow <= 0) return true; - const compactionSettings = this.settings.getGroup("compaction"); - const residualTokens = compactionContextTokens( - this.getContextUsage({ contextWindow })?.tokens ?? 0, - this.#estimateStoredContextTokens(), - ); - const fitBudget = Math.max(0, contextWindow - resolveBudgetReserveTokens(contextWindow, compactionSettings)); - return residualTokens <= fitBudget; - } - - /** - * Last-resort tiered reducer when {@link #runAutoCompaction} would otherwise - * dead-end. The summarizer cut at the only available turn boundary, but the - * kept tail is still over the recovery band because a single recent turn (a - * large tool-result, a heavy fenced/XML block, attached images) is itself - * bigger than the band and `findCutPoint` cannot cut inside one message. - * - * Tier 1 — `shake("elide")` reaches INSIDE that tail: heavy tool-result / - * block content is offloaded to one `artifact://` blob behind a recoverable - * placeholder. Skipped when this pass already ran a shake (`skipElide`). - * Tier 2 — `dropImages()`: the manual `/shake images` remedy, automated. - * Image blocks are stripped from the branch; unlike elided text they are NOT - * artifact-recoverable, so this tier only runs once elide has failed the - * progress re-test. - * - * Each tier that rewrote history re-anchors the in-flight context snapshot, - * then the caller's progress predicate is re-tested; the first tier that - * restores progress emits one info notice describing everything freed and - * stops. Returns whether progress was restored — `false` falls through to - * the dead-end warning. - */ - async #rescueCompactionDeadEnd( - signal: AbortSignal, - options: { skipElide: boolean; hasProgress: () => boolean }, - ): Promise { - if (signal.aborted) return false; - // Tier 0 — a snapcompact pass whose just-written frame archive is itself - // the over-budget cost (each pass re-renders the carried-forward text - // into MORE frames, so the archive grows past the recovery band and the - // elide/image tiers below can never shrink it): rebuild the archive at - // a threshold-derived frame budget. - const frameRescue = await this.#rescueSnapcompactFrameOverflow( - this.sessionManager.getBranch(), - this.settings.getGroup("compaction"), - signal, - ); - if (frameRescue !== undefined && options.hasProgress()) return true; - let elided = 0; - let elidedTokens = 0; - let elideSink = "placeholders"; - if (!options.skipElide) { - try { - const result = await this.shake("elide", { signal }); - elided = result.toolResultsDropped + result.blocksDropped; - elidedTokens = result.tokensFreed; - if (result.artifactId) elideSink = "an artifact"; - if (elided > 0) { - // The elide pass rewrote history; re-anchor the in-flight snapshot - // so the caller's headroom/retry-fit re-test measures the shaken - // context. - this.#rebasePendingContextSnapshotAfterCompaction(); - } - } catch (error) { - logger.warn("Dead-end shake rescue failed", { - error: error instanceof Error ? error.message : String(error), - }); - } - if (elided > 0 && options.hasProgress()) { - this.emitNotice( - "info", - `Compaction dead-end recovery: ${this.#describeElideRescue(elided, elidedTokens, elideSink)} so maintenance could make progress.`, - "compaction", - ); - return true; - } - } - if (signal.aborted) return false; - let imagesDropped = 0; - try { - imagesDropped = (await this.dropImages()).removed; - if (imagesDropped > 0) this.#rebasePendingContextSnapshotAfterCompaction(); - } catch (error) { - logger.warn("Dead-end image-drop rescue failed", { - error: error instanceof Error ? error.message : String(error), - }); - } - if (imagesDropped > 0 && options.hasProgress()) { - const elidedPart = elided > 0 ? `${this.#describeElideRescue(elided, elidedTokens, elideSink)} and ` : ""; - this.emitNotice( - "info", - `Compaction dead-end recovery: ${elidedPart}dropped ${imagesDropped} attached image${imagesDropped === 1 ? "" : "s"} so maintenance could make progress.`, - "compaction", - ); - return true; - } - return false; - } - - /** Notice fragment for a dead-end elide tier: what was freed and where it went. */ - #describeElideRescue(elided: number, tokensFreed: number, sink: string): string { - return `elided ${elided} heavy block${elided === 1 ? "" : "s"} (~${tokensFreed.toLocaleString()} tokens) to ${sink}`; - } - - /** - * Frame budget for {@link #rescueSnapcompactFrameOverflow}: targets - * `COMPACTION_RECOVERY_BAND × threshold` (the same band - * {@link #compactionCreatedHeadroom} re-tests), not the window-fit budget - * {@link #computeSnapcompactMaxFrames} sizes against — a rebuilt archive - * must land back under the maintenance trigger, or the next settle - * re-enters the same dead-end. Cap reserve mirrors - * #computeSnapcompactMaxFrames (text edges + summary template), and - * `keptTailTokens` charges the kept entries AFTER the archive so the - * budget mirrors what #compactionCreatedHeadroom will actually measure. - * Returns 0 when not even one frame fits that budget — the rebuild could - * never create headroom, so the caller must not append it. - */ - #computeSnapcompactRescueMaxFrames(settings: CompactionSettings, keptTailTokens: number): number { - const ctxWindow = this.model?.contextWindow ?? 0; - if (ctxWindow <= 0) return Math.min(snapcompact.MAX_FRAMES_DEFAULT, snapcompact.maxFramesForDataBudget()); - const thresholdTokens = resolveThresholdTokens(ctxWindow, settings); - const recoveryBandTokens = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); - const baseTokens = computeNonMessageTokens(this); - const shape = snapcompact.resolveShape(this.model, this.settings.get("snapcompact.shape")); - const edgeCap = snapcompact.geometry(shape).capacity; - const textEdgeTokens = Math.ceil((2 * edgeCap * 1.15) / 4); - const SUMMARY_TEMPLATE_TOKENS = 2000; - const frameBudget = recoveryBandTokens - baseTokens - keptTailTokens - textEdgeTokens - SUMMARY_TEMPLATE_TOKENS; - if (frameBudget < snapcompact.FRAME_TOKEN_ESTIMATE) return 0; - // Same hard caps as #computeSnapcompactMaxFrames: a threshold-derived - // count above the per-request payload budget would "shrink" a huge - // archive to a frame count the rebuilt prompt can never attach anyway. - return Math.min( - Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), - snapcompact.MAX_FRAMES_DEFAULT, - snapcompact.maxFramesForDataBudget(), - ); - } - - /** - * Dead-end rescue for a branch whose latest snapcompact CompactionEntry is - * itself billed past the maintenance threshold - * (`FRAME_TOKEN_ESTIMATE × frames`). Reaching the `!preparation` dead-end - * proves everything after that entry is already kept-recent (nothing to - * summarize), so the archive is the irreducible cost — and the elide/image - * tiers can never touch it: `collectShakeRegions` and `dropImages()` only - * inspect "message"/"custom_message" entries, so a `type: "compaction"` - * entry falls through both and the session re-warns on every resume (the - * shape issue #4786's rescue does not cover). - * - * Rebuilds the SAME archive locally — no LLM, no network — by re-running - * `snapcompact.compact()` over the entry's carried-forward source text at - * a maxFrames derived from the trigger threshold instead of the window: - * `planArchive` truncates the oldest chars to fit, so the rebuilt entry - * genuinely shrinks. The rebuilt entry keeps the stale entry's - * `firstKeptEntryId`, so the kept tail is untouched, and persisting - * through `appendCompaction()` lets the write-time superseded-compaction - * elision drop the stale frame payload from the JSONL automatically. - */ - async #rescueSnapcompactFrameOverflow( - branchEntries: SessionEntry[], - settings: CompactionSettings, - signal: AbortSignal, - ): Promise { - if (signal.aborted) return undefined; - // Re-rendering frames needs a vision-capable model, same gate as the - // snapcompact strategy path. - if (!this.model?.input.includes("image")) return undefined; - const staleEntry = getLatestCompactionEntry(branchEntries); - if (!staleEntry) return undefined; - // Only rescue when the archive is the actual source of the overflow. - // The frame budget below charges every kept entry the rebuilt context - // will still carry — the kept-recent region from `firstKeptEntryId` - // (re-emitted before the archive by buildSessionContext) plus the - // entries after the archive — on top of the fixed context, mirroring - // what #compactionCreatedHeadroom will measure. When not even one - // frame fits (e.g. a huge kept tool result dominates), rebuilding - // would append the replacement compaction at the leaf — turning the - // branch tail into a compaction entry, which prepareCompaction's - // last-entry guard can never summarize past even after an elide - // shrinks the real culprit. Bail and let the elide/image tiers handle - // that tail instead. - let keptTailTokens = 0; - let inKeptRegion = false; - for (const entry of branchEntries) { - if (entry.id === staleEntry.firstKeptEntryId) inKeptRegion = true; - if (entry.id === staleEntry.id) { - // Everything after the archive is always kept. - inKeptRegion = true; - continue; - } - if (!inKeptRegion) continue; - const message = (entry as { message?: AgentMessage }).message; - if (message) keptTailTokens += estimateTokens(message); - } - const archive = snapcompact.getPreservedArchive(staleEntry.preserveData); - if (!archive || archive.frames.length <= 1) return undefined; - const archiveText = snapcompact.archiveSourceText(archive); - if (!archiveText) return undefined; - const maxFrames = this.#computeSnapcompactRescueMaxFrames(settings, keptTailTokens); - if (maxFrames < 1 || maxFrames >= archive.frames.length) return undefined; - - const staleDetails = staleEntry.details as snapcompact.CompactionDetails | undefined; - const fileOps = snapcompact.createFileOps(); - for (const file of staleDetails?.readFiles ?? []) fileOps.read.add(file); - for (const file of staleDetails?.modifiedFiles ?? []) fileOps.edited.add(file); - const shapeSetting = this.settings.get("snapcompact.shape"); - const shape = snapcompact.resolveShapeForText(archiveText, this.model, shapeSetting); - let result: snapcompact.CompactionResult; - try { - result = await snapcompact.compact( - { - firstKeptEntryId: staleEntry.firstKeptEntryId, - messagesToSummarize: [], - turnPrefixMessages: [], - tokensBefore: staleEntry.tokensBefore, - previousSummary: staleEntry.summary, - previousPreserveData: staleEntry.preserveData, - fileOps, - }, - { - convertToLlm, - model: this.model, - ...(shapeSetting === "auto" ? {} : { shape }), - maxFrames, - }, - ); - } catch (error) { - logger.warn("Dead-end snapcompact frame rescue failed", { - error: error instanceof Error ? error.message : String(error), - }); - return undefined; - } - if (signal.aborted) return undefined; - const rebuilt = snapcompact.getPreservedArchive(result.preserveData); - if (!rebuilt || rebuilt.frames.length >= archive.frames.length) return undefined; - - const rebuiltEntryId = this.sessionManager.appendCompaction( - result.summary, - result.shortSummary, - result.firstKeptEntryId, - result.tokensBefore, - result.details, - false, - result.preserveData, - ); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#rebasePendingContextSnapshotAfterCompaction(); - // Same post-rewrite bookkeeping as the regular compaction append: the - // rebuilt context no longer carries the transient plan reference (#1246), - // and advisor cursors / todo phases were derived from the replaced - // history. - this.#planReferenceSent = false; - this.#resetAllAdvisorRuntimes(); - this.#syncTodoPhasesFromBranch(); - this.#closeCodexProviderSessionsForHistoryRewrite(); - // Extensions must see the entry that is now active, not (only) the one - // this rebuild just superseded — mirror the regular append path's hook. - const rebuiltEntry = this.sessionManager.getEntries().find(e => e.id === rebuiltEntryId) as - | CompactionEntry - | undefined; - if (this.#extensionRunner && rebuiltEntry) { - await this.#extensionRunner.emit({ - type: "session_compact", - compactionEntry: rebuiltEntry, - fromExtension: false, - }); - } - this.emitNotice( - "info", - `Compaction dead-end recovery: rebuilt the trailing snapcompact archive at a smaller frame budget (${archive.frames.length} → ${rebuilt.frames.length} frames) so maintenance could make progress.`, - "compaction", - ); - return result; - } - - /** - * Internal: Run auto-compaction with events. - * - * @param allowDefer If true (default), threshold-driven handoff strategy is allowed to - * schedule itself as a deferred post-prompt task and return a deferred-handoff result - * immediately. The caller MUST treat that as "compaction will happen async — do not - * also schedule `agent.continue()` for this turn", otherwise the deferred handoff - * races a fresh streaming turn (the symptom: "Auto-handoff" loader + assistant - * message still streaming). Callers on a path that is about to start a new agent - * turn (e.g. the pre-prompt check in `#promptWithMessage`) pass `false` to force - * inline execution so the handoff completes before the new turn begins. - * @returns whether auto-compaction scheduled a follow-up turn. - */ - async #runAutoCompaction( - reason: "overflow" | "threshold" | "idle" | "incomplete", - willRetry: boolean, - deferred = false, - allowDefer = true, - options: { - autoContinue?: boolean; - triggerContextTokens?: number; - suppressContinuation?: boolean; - suppressHandoff?: boolean; - phase?: CodexCompactionContext["phase"]; - terminalTextAnswer?: boolean; - } = {}, - ): Promise { - const compactionSettings = this.settings.getGroup("compaction"); - if (compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; - if (reason !== "idle" && !compactionSettings.enabled) return COMPACTION_CHECK_NONE; - const generation = this.#promptGeneration; - const terminalTextAnswer = - options.terminalTextAnswer ?? isTerminalTextAssistantAnswer(this.#findLastAssistantMessage()); - const suppressContinuation = options.suppressContinuation === true; - const shouldAutoContinue = - !suppressContinuation && options.autoContinue !== false && compactionSettings.autoContinue !== false; - const suppressHandoff = options.suppressHandoff === true; - let fallbackFromShake = false; - // Shake runs inline (cheap, no remote LLM). On overflow recovery, if shake - // reclaims nothing we fall through to the summary-compaction body below so - // the oversized input still gets resolved. - if (compactionSettings.strategy === "shake") { - const outcome = await this.#runAutoShake( - reason, - willRetry, - generation, - shouldAutoContinue, - terminalTextAnswer, - options.triggerContextTokens, - suppressContinuation, - ); - if (outcome !== "fallback") return outcome; - fallbackFromShake = true; - } - // "overflow" and "incomplete" force inline execution because they are recovery - // paths the caller wants resolved before scheduling the next turn. "idle" is - // triggered by the idle loop and does its own scheduling. - if ( - !suppressHandoff && - !deferred && - allowDefer && - reason !== "overflow" && - reason !== "incomplete" && - reason !== "idle" && - compactionSettings.strategy === "handoff" - ) { - this.#schedulePostPromptTask( - async signal => { - await Promise.resolve(); - if (signal.aborted) return; - await this.#runAutoCompaction(reason, willRetry, true, true, { - ...options, - terminalTextAnswer, - }); - }, - { generation }, - ); - return { - ...COMPACTION_CHECK_DEFERRED_HANDOFF, - continuationScheduled: shouldAutoContinue, - }; - } - - // "overflow" forces context-full because the input itself is broken — a handoff - // LLM call would hit the same overflow. "incomplete" is an output-side problem, - // so a handoff request on the existing context is still viable. - let action: "context-full" | "handoff" | "snapcompact" = - compactionSettings.strategy === "snapcompact" - ? "snapcompact" - : compactionSettings.strategy === "handoff" && reason !== "overflow" && !suppressHandoff - ? "handoff" - : "context-full"; - if (action === "snapcompact" && this.model && !this.model.input.includes("image")) { - this.emitNotice( - "warning", - `snapcompact needs a vision-capable active model (${this.model.id} is text-only); using context-full auto-compaction instead.`, - "compaction", - ); - action = "context-full"; - } - // Abort any older auto-compaction before installing this run's controller. - this.#autoCompactionAbortController?.abort(); - const autoCompactionAbortController = new AbortController(); - this.#autoCompactionAbortController = autoCompactionAbortController; - const autoCompactionSignal = autoCompactionAbortController.signal; - - try { - // Emit start AFTER the controller is installed so isCompacting is already true - // for any listener — and for input routed during this emit's event-loop yield: - // a message typed as the compaction loader appears must land in the compaction - // queue, not the core steering queue (which handoff's agent.reset() would wipe). - await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action }); - if (action === "handoff") { - let handoffSwitchCancelled = false; - const handoffFocus = AUTO_HANDOFF_THRESHOLD_FOCUS; - const handoffResult = await this.handoff(handoffFocus, { - autoTriggered: true, - signal: autoCompactionSignal, - onSwitchCancelled: () => { - handoffSwitchCancelled = true; - }, - }); - if (!handoffResult) { - const aborted = autoCompactionSignal.aborted || handoffSwitchCancelled; - if (aborted) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: true, - willRetry: false, - }); - return COMPACTION_CHECK_NONE; - } - logger.warn("Auto-handoff returned no document; falling back to context-full maintenance", { - reason, - }); - action = "context-full"; - } - if (handoffResult) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - }); - const continuationScheduled = - !autoCompactionSignal.aborted && - this.#scheduleCompactionContinuation({ - generation, - autoContinue: reason !== "idle" && shouldAutoContinue, - terminalTextAnswer, - suppressContinuation, - }); - return { - ...(continuationScheduled ? COMPACTION_CHECK_CONTINUATION : COMPACTION_CHECK_NONE), - historyRewritten: true, - }; - } - } - - if (!this.model) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - skipped: true, - }); - return COMPACTION_CHECK_NONE; - } - - const availableModels = this.#modelRegistry.getAvailable(); - if (availableModels.length === 0) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - skipped: true, - }); - return COMPACTION_CHECK_NONE; - } - - const pathEntries = this.sessionManager.getBranch(); - - let pathEntriesForCompaction = pathEntries; - let preparation = prepareCompaction(pathEntriesForCompaction, compactionSettings, this.model); - if (!preparation) { - // prepareCompaction found nothing to summarize because the kept region - // is a single oversized recent turn — findCutPoint never cuts inside a - // tool result, so a huge tool-result / fenced block tail leaves nothing - // on the summarizable side and summary compaction cannot even start. - // That is exactly the dead-end the elide shake rescues: it reaches - // INSIDE the tail and offloads heavy content to an artifact placeholder, - // shrinking the tail so findCutPoint can then move the cut and leave - // older turns to summarize. Run the same tiered rescue the - // post-maintenance guard uses (elide, then image drop), with progress - // defined as "prepareCompaction now succeeds on the rewritten branch", - // and fall through to the normal compaction body when it does (writing - // a compaction entry anchors the stale billed usage so the - // auto-continue re-check cannot re-trip and loop the warning — issue - // #4786). `skipElide` when we already fell through from a shake - // strategy pass (it tried and found nothing); skip entirely on the - // idle timer (it re-checks usage on its own cadence). - let rescueRewroteHistory = false; - // A snapcompact CompactionEntry is invisible to both rescue tiers - // below (they only inspect message entries) and to prepareCompaction - // itself (last-entry-is-compaction guard), so a frame archive billed - // past the threshold dead-ends here on every resume. Rebuild it at a - // threshold-derived frame budget first — but treat that as complete - // only when it actually created headroom: the latest archive may not - // be the oversized tail (e.g. a huge kept tool result after it), and - // declaring victory on a mere frame-count shrink would skip the - // elide/image tiers that can still reach that tail and suppress a - // warning the user should see. - let frameRescueResult: snapcompact.CompactionResult | undefined; - let frameRescueCreatedHeadroom = false; - if (reason !== "idle") { - frameRescueResult = await this.#rescueSnapcompactFrameOverflow( - pathEntriesForCompaction, - compactionSettings, - autoCompactionSignal, - ); - if (frameRescueResult) { - rescueRewroteHistory = true; - pathEntriesForCompaction = this.sessionManager.getBranch(); - frameRescueCreatedHeadroom = this.#compactionCreatedHeadroom(); - } - if (!frameRescueCreatedHeadroom) { - await this.#rescueCompactionDeadEnd(autoCompactionSignal, { - skipElide: fallbackFromShake, - hasProgress: () => { - // Only reached when a tier actually freed something, so the - // branch has been rewritten either way. - rescueRewroteHistory = true; - pathEntriesForCompaction = this.sessionManager.getBranch(); - preparation = prepareCompaction(pathEntriesForCompaction, compactionSettings, this.model); - return preparation !== undefined; - }, - }); - } - } - if (!preparation) { - const noProgressDeadEnd = reason !== "idle" && !frameRescueCreatedHeadroom; - const deadEndWarning = noProgressDeadEnd - ? compactionDeadEndWarning("shrink it (e.g. clear large tool output)") - : undefined; - // A rescue that appended a rebuilt archive without creating - // headroom must carry the dead-end badge on the entry the - // transcript actually shows (the rebuilt one), or the pause - // loses its explanation once the notice scrolls away. Stamp it - // BEFORE the auto_compaction_end event: a result-carrying event - // makes the TUI rebuild the chat from the current entries - // immediately, so a later stamp would not appear until some - // unrelated rebuild. - if (deadEndWarning && frameRescueResult) { - const stampEntry = getLatestCompactionEntry(this.sessionManager.getBranch()); - if (stampEntry) { - stampEntry.warning = deadEndWarning; - await this.sessionManager.rewriteEntries(); - } - } - // A successful frame rescue rewrote history and activated a new - // compaction entry — surface it as a real (non-skipped) result so - // the TUI rebuilds the transcript instead of treating the pass as - // a benign no-op. - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: frameRescueResult, - aborted: false, - willRetry: false, - skipped: frameRescueResult === undefined, - }); - let continuationScheduled = false; - if (frameRescueCreatedHeadroom) { - continuationScheduled = this.#scheduleCompactionContinuation({ - generation, - autoContinue: shouldAutoContinue, - terminalTextAnswer, - suppressContinuation, - }); - } else if (!suppressContinuation && this.agent.hasQueuedMessages()) { - this.#scheduleAgentContinue({ - delayMs: 100, - generation, - shouldContinue: () => this.agent.hasQueuedMessages(), - }); - continuationScheduled = true; - } - if (deadEndWarning) { - this.emitNotice("warning", deadEndWarning, "compaction"); - } - // A rescue that offloaded content but still could not produce a - // preparation rewrote the branch; flag it so the overflow-recovery - // rollback does not re-restore the just-failed assistant turn on top - // of the elided tail. - const base = continuationScheduled - ? COMPACTION_CHECK_CONTINUATION - : noProgressDeadEnd - ? COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION - : COMPACTION_CHECK_NONE; - return rescueRewroteHistory ? { ...base, historyRewritten: true } : base; - } - } - - let hookCompaction: CompactionResult | undefined; - let fromExtension = false; - let preserveData: Record | undefined; - let codexCompaction: CodexCompactionContext | undefined; - - if (this.#extensionRunner?.hasHandlers("session_before_compact")) { - const hookResult = (await this.#extensionRunner.emit({ - type: "session_before_compact", - preparation, - branchEntries: pathEntriesForCompaction, - customInstructions: undefined, - signal: autoCompactionSignal, - })) as SessionBeforeCompactResult | undefined; - - if (hookResult?.cancel) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: true, - willRetry: false, - }); - return COMPACTION_CHECK_NONE; - } - - if (hookResult?.compaction) { - hookCompaction = hookResult.compaction; - fromExtension = true; - } - } - - const compactionPrep = await this.#prepareCompactionFromHooks(preparation, hookCompaction); - - let summary: string; - let shortSummary: string | undefined; - let firstKeptEntryId: string; - let tokensBefore: number; - let details: unknown; - - // Snapcompact runs locally first. The post-compaction context = kept-recent - // + a summary message carrying the imaged archive at FRAME_TOKEN_ESTIMATE - // per frame; #computeSnapcompactMaxFrames sizes the frame cap from the - // live window so we don't run snapcompact just to overflow every threshold - // tick. Any local blocker (unsupported snapcompact glyphs, kept-history too large, - // post-render overflow) downgrades auto maintenance to a context-full LLM - // summary instead of wedging the session (#3659) — auto runs the default - // strategy on the user's behalf, so a fallback that lets the session keep - // running is the right behavior. Manual `/compact snapcompact` keeps the - // local-only contract (#3599): the user explicitly picked it. - let snapcompactResult: snapcompact.CompactionResult | undefined; - let snapcompactBlocker: string | undefined; - if (action === "snapcompact" && compactionPrep.kind !== "fromHook") { - // Drop `¶think:` sections for Anthropic-dialect targets: the archive - // is replayed as text and Claude refuses reproduced reasoning - // ("reasoning_extraction", issue #6093). - const snapcompactIncludeThinking = preferredDialect(this.model.id) !== "anthropic"; - const text = snapcompact.serializeConversation( - convertToLlm(preparation.messagesToSummarize.concat(preparation.turnPrefixMessages)), - { includeThinking: snapcompactIncludeThinking }, - ); - const probeText = snapcompact.renderabilityProbeText( - text, - preparation.previousPreserveData, - preparation.previousSummary, - ); - const shapeSetting = this.settings.get("snapcompact.shape"); - const shape = snapcompact.resolveShapeForText(probeText, this.model, shapeSetting); - const renderScan = snapcompact.scanRenderability(probeText, { shape }); - if (!renderScan.isSafe) { - const percent = (renderScan.unrenderableRatio * 100).toFixed(1); - logger.warn("Snapcompact disabled: unsupported characters for selected snapcompact font", { - model: this.model?.id, - unrenderableRatio: renderScan.unrenderableRatio, - }); - snapcompactBlocker = `snapcompact disabled: unsupported characters for selected snapcompact font (${percent}%); using context-full auto-compaction instead.`; - } else { - const maxFrames = this.#computeSnapcompactMaxFrames(preparation, compactionSettings); - if (maxFrames < 1) { - logger.warn("Snapcompact skipped: kept history alone exceeds the context budget", { - model: this.model?.id, - }); - snapcompactBlocker = - "snapcompact: kept history alone exceeds the context budget; using context-full auto-compaction instead."; - } else { - snapcompactResult = await snapcompact.compact(preparation, { - convertToLlm, - model: this.model, - ...(shapeSetting === "auto" ? {} : { shape }), - maxFrames, - includeThinking: snapcompactIncludeThinking, - }); - const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); - if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { - logger.warn("Snapcompact exceeded the per-request frame payload budget", { - model: this.model?.id, - framePayloadBytes, - budget: snapcompact.FRAME_DATA_BYTES_BUDGET, - }); - snapcompactBlocker = - "snapcompact produced too much standing image payload; using context-full auto-compaction instead."; - snapcompactResult = undefined; - } - if (snapcompactResult) { - const ctxWindow = this.model?.contextWindow ?? 0; - const budget = - ctxWindow > 0 - ? ctxWindow - effectiveReserveTokens(ctxWindow, compactionSettings) - : Number.POSITIVE_INFINITY; - const projected = this.#projectSnapcompactContextTokens(preparation, snapcompactResult); - if (projected > budget) { - logger.warn("Snapcompact still overflows the window after frame-budget sizing", { - model: this.model?.id, - projected, - budget, - }); - snapcompactBlocker = - "snapcompact could not bring the context under the limit; using context-full auto-compaction instead."; - snapcompactResult = undefined; - } - } - } - } - if (snapcompactBlocker) { - this.emitNotice("warning", snapcompactBlocker, "compaction"); - action = "context-full"; - } - } - - if (compactionPrep.kind === "fromHook") { - summary = compactionPrep.summary; - shortSummary = compactionPrep.shortSummary; - firstKeptEntryId = compactionPrep.firstKeptEntryId; - tokensBefore = compactionPrep.tokensBefore; - details = compactionPrep.details; - preserveData = compactionPrep.preserveData; - } else if (snapcompactResult) { - summary = snapcompactResult.summary; - shortSummary = snapcompactResult.shortSummary; - firstKeptEntryId = snapcompactResult.firstKeptEntryId; - tokensBefore = snapcompactResult.tokensBefore; - details = snapcompactResult.details; - preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) }; - } else { - const candidates = this.#getCompactionModelCandidates(availableModels); - const retrySettings = this.settings.getGroup("retry"); - const telemetry = resolveTelemetry(this.agent.telemetry, this.sessionId); - let compactResult: CompactionResult | undefined; - let lastError: unknown; - codexCompaction = createCodexCompactionContext({ - trigger: "auto", - reason: "context_limit", - phase: - options.phase ?? - (reason === "threshold" ? "pre_turn" : reason === "idle" ? "standalone_turn" : "mid_turn"), - }); - - for (let candidateIndex = 0; candidateIndex < candidates.length; candidateIndex++) { - const candidate = candidates[candidateIndex]; - const hasMoreCandidates = candidateIndex < candidates.length - 1; - const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId); - if (!apiKey) continue; - - let attempt = 0; - while (true) { - try { - compactResult = await compact( - this.#obfuscatePreparationForProvider(preparation), - candidate, - this.#modelRegistry.resolver(candidate, this.sessionId), - undefined, - autoCompactionSignal, - { - promptOverride: this.#obfuscateTextForProvider(compactionPrep.hookPrompt), - extraContext: compactionPrep.hookContext, - remoteInstructions: this.#baseSystemPrompt.join("\n\n"), - metadata: this.agent.metadataForProvider(candidate.provider), - initiatorOverride: "agent", - convertToLlm: messages => this.#convertToLlmForSideRequest(messages), - telemetry, - // Honor the user's /model thinking selection on the - // auto-compaction path — the most-fired compaction - // site. Clamped per-model inside compact() via - // resolveCompactionEffort. - thinkingLevel: this.thinkingLevel, - tools: this.agent.state.tools, - sessionId: this.sessionId, - promptCacheKey: this.sessionId, - providerSessionState: this.#providerSessionState, - codexCompaction, - }, - ); - break; - } catch (error) { - if (autoCompactionSignal.aborted) { - throw error; - } - - const message = error instanceof Error ? error.message : String(error); - const id = AIError.classify(error, candidate.api); - if (AIError.is(id, AIError.Flag.AuthFailed)) { - lastError = this.#buildCompactionAuthError(); - break; - } - if (AIError.is(id, AIError.Flag.Timeout)) { - logger.warn( - hasMoreCandidates - ? "Auto-compaction summarization timed out, trying next model" - : "Auto-compaction summarization timed out, not retrying same model", - { - error: message, - model: `${candidate.provider}/${candidate.id}`, - }, - ); - lastError = error; - break; - } - - const retryAfterMs = this.#parseRetryAfterMsFromError(message); - const shouldRetry = - retrySettings.enabled && - attempt < retrySettings.maxRetries && - (retryAfterMs !== undefined || - AIError.is(id, AIError.Flag.Transient) || - AIError.is(id, AIError.Flag.UsageLimit)); - if (!shouldRetry) { - lastError = error; - break; - } - - const baseDelayMs = retrySettings.baseDelayMs * 2 ** attempt; - const delayMs = retryAfterMs !== undefined ? Math.max(baseDelayMs, retryAfterMs) : baseDelayMs; - - // If retry delay is too long (>30s), try next candidate instead of waiting - const maxAcceptableDelayMs = 30_000; - if (delayMs > maxAcceptableDelayMs && hasMoreCandidates) { - logger.warn("Auto-compaction retry delay too long, trying next model", { - delayMs, - retryAfterMs, - error: message, - model: `${candidate.provider}/${candidate.id}`, - }); - lastError = error; - break; // Exit retry loop, continue to next candidate - } - - attempt++; - logger.warn("Auto-compaction failed, retrying", { - attempt, - maxRetries: retrySettings.maxRetries, - delayMs, - retryAfterMs, - error: message, - model: `${candidate.provider}/${candidate.id}`, - }); - await scheduler.wait(delayMs, { signal: autoCompactionSignal }); - } - } - - if (compactResult) { - break; - } - } - - if (!compactResult) { - if (lastError) { - throw lastError; - } - throw new Error("Compaction failed: no available model"); - } - - summary = compactResult.summary; - shortSummary = compactResult.shortSummary; - firstKeptEntryId = compactResult.firstKeptEntryId; - tokensBefore = compactResult.tokensBefore; - details = compactResult.details; - preserveData = mergeLlmCompactionPreserveData(compactionPrep.preserveData, compactResult.preserveData); - } - - if (autoCompactionSignal.aborted) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: true, - willRetry: false, - }); - return COMPACTION_CHECK_NONE; - } - - this.sessionManager.appendCompaction( - summary, - shortSummary, - firstKeptEntryId, - tokensBefore, - details, - fromExtension, - preserveData, - ); - const newEntries = this.sessionManager.getEntries(); - const sessionContext = this.buildDisplaySessionContext(); - this.agent.replaceMessages(sessionContext.messages); - this.#rebasePendingContextSnapshotAfterCompaction(); - // Compaction discarded the conversation history that carried the approved - // plan reference. Clear the sent-flag so #buildPlanReferenceMessage re-reads - // the plan from disk and re-injects it on the next turn (issue #1246). - this.#planReferenceSent = false; - this.#resetAllAdvisorRuntimes(); - this.#syncTodoPhasesFromBranch(); - if (codexCompaction) { - this.#resetCodexProviderAfterCompaction(codexCompaction); - } else { - this.#closeCodexProviderSessionsForHistoryRewrite(); - } - - // Get the saved compaction entry for the hook - const savedCompactionEntry = newEntries.find(e => e.type === "compaction" && e.summary === summary) as - | CompactionEntry - | undefined; - - if (this.#extensionRunner && savedCompactionEntry) { - await this.#extensionRunner.emit({ - type: "session_compact", - compactionEntry: savedCompactionEntry, - fromExtension, - }); - } - - const result: CompactionResult = { - summary, - shortSummary, - firstKeptEntryId, - tokensBefore, - details, - preserveData, - }; - // Post-maintenance progress guard — evaluated BEFORE emitting - // auto_compaction_end so the TUI rebuild triggered by that event - // already reflects any rescue rewrite (elide / image-drop) and the - // dead-end warning stamped on the compaction entry. Snapcompact can - // project over budget and fall back to a context-full summary; the - // summarizer keeps `keepRecentTokens` of recent history verbatim and - // findCutPoint can only cut at turn boundaries (never tool results), - // so a single oversized recent turn (e.g. a huge tool result) leaves - // the rewritten context still above threshold. Scheduling the - // continuation regardless means the next agent_end re-enters - // #checkCompaction over the same oversized tail and re-fires forever. - // The retry and the threshold auto-continue use different progress - // tests (a recoverable overflow only has to fit; the auto-continue - // thrash needs the stricter recovery band), so each branch evaluates - // its own below. - let continuationScheduled = false; - // A non-idle pass that wanted to continue (retry or auto-continue) but freed - // too little for that path to proceed is a dead-end: warn once so the user - // understands why maintenance paused instead of silently looping. - let noProgressDeadEnd = false; - let retryFits = false; - let hasHeadroom = false; - - if (willRetry) { - const messages = this.agent.state.messages; - const lastMsg = messages[messages.length - 1]; - if (lastMsg?.role === "assistant") { - const lastAssistant = lastMsg as AssistantMessage; - // Drop the prior turn before retry when it carries no actionable deliverable: - // - "error": failure was kept in history but must not re-enter the next turn's prompt. - // - reason === "incomplete" && stopReason === "length": truncated output (typically - // reasoning-only) — re-running it produces the same dead-end. - const shouldDrop = - lastAssistant.stopReason === "error" || - (reason === "incomplete" && lastAssistant.stopReason === "length"); - if (shouldDrop) { - this.agent.replaceMessages(messages.slice(0, -1)); - this.#rebasePendingContextSnapshotAfterCompaction(); - } - } - - // Retry only needs the rebuilt prompt to fit the window again — measured - // AFTER the drop above so the just-failed turn (which the retry prompt - // won't include) is excluded. Reusing the auto-continue recovery band - // here turned recoverable overflows into manual dead-ends (#3412 review), - // so use the looser fit budget. - retryFits = this.#compactionCreatedRetryFit(); - if (!retryFits) { - retryFits = await this.#rescueCompactionDeadEnd(autoCompactionSignal, { - skipElide: fallbackFromShake, - hasProgress: () => this.#compactionCreatedRetryFit(), - }); - } - if (!retryFits) { - noProgressDeadEnd = true; - } - } else if (reason !== "idle") { - // Mirror the shake recovery-band check: only auto-continue when compaction - // landed residual context under `COMPACTION_RECOVERY_BAND × threshold`. - // Re-firing on a history that still sits just over the line is the - // snapcompact thrash, so require genuine headroom, not a bare fit. Even - // when auto-continue is disabled, a no-headroom threshold pass must still - // block later automatic continuations (todo reminders/session_stop hooks) - // from re-entering the same oversized context. - hasHeadroom = this.#compactionCreatedHeadroom(); - if (!hasHeadroom) { - hasHeadroom = await this.#rescueCompactionDeadEnd(autoCompactionSignal, { - skipElide: fallbackFromShake, - hasProgress: () => this.#compactionCreatedHeadroom(), - }); - } - if (!hasHeadroom) { - noProgressDeadEnd = true; - } - } - - const deadEndWarning = noProgressDeadEnd ? compactionDeadEndWarning("clear large tool output") : undefined; - if (deadEndWarning) { - // Stamp the divider: the compaction bar badges the dead-end and - // carries the full warning in its ctrl+o detail, so the pause - // stays explained even after the notice row scrolls away. Stamp - // the branch's LATEST compaction entry — a frame rescue may have - // superseded `savedCompactionEntry` with a rebuilt one, and the - // collapsed transcript badges only the active entry. - const stampEntry = getLatestCompactionEntry(this.sessionManager.getBranch()) ?? savedCompactionEntry; - if (stampEntry) { - stampEntry.warning = deadEndWarning; - await this.sessionManager.rewriteEntries(); - } - } - - await this.#emitSessionEvent({ type: "auto_compaction_end", action, result, aborted: false, willRetry }); - - if (retryFits) { - this.#scheduleAgentContinue({ delayMs: 100, generation }); - continuationScheduled = true; - } else { - continuationScheduled = this.#scheduleCompactionContinuation({ - generation, - autoContinue: hasHeadroom && shouldAutoContinue, - terminalTextAnswer, - suppressContinuation, - }); - } - - if (deadEndWarning) { - this.emitNotice("warning", deadEndWarning, "compaction"); - } - if (continuationScheduled) return COMPACTION_CHECK_CONTINUATION; - return noProgressDeadEnd ? COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION : COMPACTION_CHECK_NONE; - } catch (error) { - if (autoCompactionSignal.aborted) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: true, - willRetry: false, - }); - return COMPACTION_CHECK_NONE; - } - const errorMessage = error instanceof Error ? error.message : "compaction failed"; - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - errorMessage: - reason === "overflow" - ? `Context overflow recovery failed: ${errorMessage}` - : reason === "incomplete" - ? `Incomplete response recovery failed: ${errorMessage}` - : `Auto-compaction failed: ${errorMessage}`, - }); - } finally { - if (this.#autoCompactionAbortController === autoCompactionAbortController) { - this.#autoCompactionAbortController = undefined; - } - } - return COMPACTION_CHECK_NONE; - } - - /** - * Run a shake-strategy auto-maintenance pass. Emits the - * `auto_compaction_start`/`auto_compaction_end` pair with a shake `action`, - * runs {@link shake} inline against the protect-window config, and schedules - * continuation exactly like the context-full tail. - * - * Returns `"fallback"` only for an overflow recovery where shake reclaimed - * nothing (or threw) — the caller then runs the summary-compaction body so - * the oversized input still gets resolved. Returns `"handled"` otherwise. - */ - async #runAutoShake( - reason: "overflow" | "threshold" | "idle" | "incomplete", - willRetry: boolean, - generation: number, - autoContinue: boolean, - terminalTextAnswer: boolean, - triggerContextTokens?: number, - suppressContinuation = false, - ): Promise { - const action = "shake"; - this.#autoCompactionAbortController?.abort(); - const controller = new AbortController(); - this.#autoCompactionAbortController = controller; - const signal = controller.signal; - try { - await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action }); - const result = await this.shake("elide", { config: DEFAULT_SHAKE_CONFIG, signal }); - if (signal.aborted) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: true, - willRetry: false, - }); - return COMPACTION_CHECK_NONE; - } - const reclaimed = result.toolResultsDropped + result.blocksDropped > 0; - // Detect the dead-loop reported in issues #2119/#2275: the threshold check - // fires, shake runs, but residual context is still above the configured - // threshold. The next agent_end would re-trigger shake, which has nothing - // new to drop on the second pass, so the loop spins until the user kills it. - // Same hazard for "incomplete" (the retry would re-hit the length cap) and - // for the existing "overflow + nothing reclaimed" case. In every recovery - // reason we hand off to the summarization-driven context-full path so the - // situation actually resolves; "idle" is exempt because its 60s+ timer - // re-checks usage before re-firing and cannot dead-loop on its own. - // - // #2275: the post-shake check MUST stay provider-anchored when caller - // usage and local estimates diverge. The local estimator undercounts - // thinking-signature payloads, so thinking-heavy sessions can read well - // below the provider usage that fired the threshold. Prefer the caller's - // context figure when supplied, then subtract shake's own savings and add - // hysteresis (80% recovery band) so we don't oscillate at the boundary. - // Threshold callers pass the provider-billed trigger after accounting for - // any supersede/drop-useless pruning that already rewrote the next prompt; - // without that pre-shake savings, shake can fall through to context-full - // even though the post-prune history is already inside the recovery band. - const contextWindow = this.model?.contextWindow ?? 0; - const compactionSettings = this.settings.getGroup("compaction"); - let stillOverThreshold = false; - if (contextWindow > 0) { - if (typeof triggerContextTokens === "number" && Number.isFinite(triggerContextTokens)) { - const correctedTokens = Math.max(0, triggerContextTokens - result.tokensFreed); - const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); - const recoveryBand = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); - stillOverThreshold = correctedTokens > recoveryBand; - } else { - const postShakeTokens = this.getContextUsage({ contextWindow })?.tokens ?? 0; - stillOverThreshold = shouldCompact(postShakeTokens, contextWindow, compactionSettings); - } - } - const shouldFallBack = reason !== "idle" && ((reason === "overflow" && !reclaimed) || stillOverThreshold); - if (shouldFallBack) { - const errorMessage = reclaimed - ? `Auto-shake reclaimed ~${result.tokensFreed} tokens but context is still above the threshold; falling back to context-full compaction.` - : "Auto-shake found nothing eligible to drop; falling back to context-full compaction."; - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - skipped: !reclaimed, - errorMessage, - }); - return "fallback"; - } - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry, - skipped: !reclaimed, - }); - - let continuationScheduled = false; - if (willRetry) { - // The shake rebuild replays every entry, so a trailing error/length - // assistant from the failed turn re-enters agent state — drop it before - // retrying, same as the context-full tail. - const messages = this.agent.state.messages; - const lastMsg = messages[messages.length - 1]; - if (lastMsg?.role === "assistant") { - const lastAssistant = lastMsg as AssistantMessage; - const shouldDrop = - lastAssistant.stopReason === "error" || - (reason === "incomplete" && lastAssistant.stopReason === "length"); - if (shouldDrop) this.agent.replaceMessages(messages.slice(0, -1)); - } - this.#scheduleAgentContinue({ delayMs: 100, generation }); - continuationScheduled = true; - } else { - continuationScheduled = this.#scheduleCompactionContinuation({ - generation, - autoContinue: reason !== "idle" && autoContinue, - terminalTextAnswer, - suppressContinuation, - }); - } - if (!reclaimed) { - return willRetry && continuationScheduled - ? { ...COMPACTION_CHECK_CONTINUATION, historyRewritten: true } - : continuationScheduled - ? COMPACTION_CHECK_CONTINUATION - : COMPACTION_CHECK_NONE; - } - return { - ...(continuationScheduled ? COMPACTION_CHECK_CONTINUATION : COMPACTION_CHECK_NONE), - historyRewritten: true, - }; - } catch (error) { - if (signal.aborted) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: true, - willRetry: false, - }); - return COMPACTION_CHECK_NONE; - } - const message = error instanceof Error ? error.message : "shake failed"; - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - errorMessage: message, - skipped: false, - }); - // Overflow still needs recovery even if shake threw. - return reason === "overflow" ? "fallback" : COMPACTION_CHECK_NONE; - } finally { - if (this.#autoCompactionAbortController === controller) { - this.#autoCompactionAbortController = undefined; - } - } - } - - /** - * Toggle auto-compaction setting. - */ - setAutoCompactionEnabled(enabled: boolean): void { - this.settings.set("compaction.enabled", enabled); - if (enabled && this.settings.get("compaction.strategy") === "off") { - const defaultStrategy = getDefault("compaction.strategy"); - this.settings.set("compaction.strategy", defaultStrategy === "off" ? "context-full" : defaultStrategy); - } - } - - /** Whether auto-compaction is enabled */ - get autoCompactionEnabled(): boolean { - return this.settings.get("compaction.enabled") && this.settings.get("compaction.strategy") !== "off"; - } - // ========================================================================= // Auto-Retry // ========================================================================= - /** - * Classify retry decisions against the active session model. Test stream - * shims and provider adapters can emit generic assistant metadata, but retry - * policy belongs to the model that was actually requested for this turn. - */ - #classifyRetryMessage(message: AssistantMessage): number { - const activeModel = this.model; - if (!activeModel || message.api === activeModel.api) { - return AIError.classifyMessage(message); - } - - const id = AIError.classifyMessage({ - api: activeModel.api, - errorId: message.errorId, - errorMessage: message.errorMessage, - errorStatus: message.errorStatus, - }); - message.errorId = id; - return id; - } - - #isGenericAbortSentinel(message: AssistantMessage): boolean { - return message.errorMessage === "Request was aborted" || message.errorMessage === "Request was aborted."; - } - - /** - * Retry an empty, reason-less provider abort: a turn with no content that - * carries the generic sentinel (bare `abort()`), whether the provider - * finalized it as `stopReason: "aborted"` or leaked it as `stopReason: - * "error"` (a stalled/dropped stream reported as an error rather than an - * abort — issue #5375). Only fires while the session is neither aborting nor - * tearing down. A user/lifecycle abort (`#abortInProgress`), a dispose-driven - * abort (`#isDisposed`), or a session-induced streaming-edit guard abort - * (`#streamingEditAbortTriggered` — auto-generated-file guard or failed-patch - * preview) is deliberate and MUST settle the turn instead: routing it through - * retry would orphan `#retryPromise` on a continuation the guard skips - * (hanging the in-flight `prompt()`) or silently undo the guard's intended - * abort. Deliberate user interrupts (`UserInterrupt`) and silent aborts carry - * their own marker, not the generic sentinel, so they never match here. - */ - #isRetryableReasonlessAbort(message: AssistantMessage): boolean { - if ( - (message.stopReason !== "aborted" && message.stopReason !== "error") || - message.content.length !== 0 || - this.#abortInProgress || - this.#isDisposed || - this.#streamingEditAbortTriggered - ) { - return false; - } - - const id = this.#classifyRetryMessage(message); - if (message.stopReason === "aborted" && AIError.is(id, AIError.Flag.Abort)) return true; - if (!this.#isGenericAbortSentinel(message)) return false; - - message.errorId = AIError.create(AIError.Flag.Abort); - return true; - } - - /** - * Check if an error is retryable (transient errors or usage limits). - * Context overflow is NOT retryable (handled by compaction instead). - * Usage-limit errors are retryable because the retry handler performs credential switching. - */ - #isRetryableError(message: AssistantMessage): boolean { - if (message.stopReason !== "error") return false; - - const id = this.#classifyRetryMessage(message); - // Context overflow is handled by compaction, not retry - const contextWindow = this.model?.contextWindow ?? 0; - if (AIError.isContextOverflow(message, contextWindow)) return false; - - if (this.#isClassifierRefusal(message)) return true; - return AIError.retriable(id, { replayUnsafe: this.#hasReplayUnsafeToolOutput(message) }); - } - - /** - * Resume a stalled turn after every emitted tool call has produced a result. - * Cursor calls must also carry the server-execution marker. The failed - * assistant/tool-result pair stays in context so completed side effects are - * continued from rather than replayed. - */ - #canResumeResolvedStreamStall(message: AssistantMessage): boolean { - if (message.stopReason !== "error" || !message.errorMessage?.toLowerCase().includes("stream stall")) { - return false; - } - const id = this.#classifyRetryMessage(message); - if (!AIError.retriable(id)) return false; - - const resolvedToolCallIds: string[] = []; - for (const block of message.content) { - if (block.type !== "toolCall") continue; - if ( - message.provider === "cursor" && - (!(kCursorExecResolved in block) || block[kCursorExecResolved] !== true) - ) { - return false; - } - resolvedToolCallIds.push(block.id); - } - if (resolvedToolCallIds.length === 0) return false; - - const messages = this.agent.state.messages; - let assistantIndex = -1; - for (let i = messages.length - 1; i >= 0; i--) { - const candidate = messages[i]; - if (candidate.role === "assistant" && this.#isSameAssistantMessage(candidate, message)) { - assistantIndex = i; - break; - } - } - if (assistantIndex < 0) return false; - - const unresolvedToolCallIds = new Set(resolvedToolCallIds); - for (let i = assistantIndex + 1; i < messages.length; i++) { - const candidate = messages[i]; - if (candidate.role === "toolResult") unresolvedToolCallIds.delete(candidate.toolCallId); - } - return unresolvedToolCallIds.size === 0; - } - /** - * Retried turns remove the failed assistant message from active context. - * Text/thinking-only partials are safe to discard and replay. Retained - * tool calls are not: a completed tool call may already have emitted its - * tool result after this assistant message, so replaying can duplicate work. - */ - #hasReplayUnsafeToolOutput(message: AssistantMessage): boolean { - return message.content.some(block => block.type === "toolCall"); - } - - /** - * OpenRouter can repeatedly close Gemini streams at the reasoning-to-payload - * transition. One retry covers a transient edge failure; the normal ten-retry - * budget would otherwise re-run the same expensive reasoning cycle unchanged. - */ - #isOpenRouterThinkingStreamClose(message: AssistantMessage): boolean { - return ( - message.provider === "openrouter" && - /server_error:\s*stream closed with reason:\s*error/i.test(message.errorMessage ?? "") && - message.content.some(block => block.type === "thinking" && block.thinking.trim().length > 0) - ); - } - - #isClassifierRefusal(message: AssistantMessage): boolean { - if (message.stopReason !== "error") return false; - const stopType = message.stopDetails?.type; - return stopType === "refusal" || stopType === "sensitive"; - } - - /** - * True when `provider` has registered models or is configured for dynamic - * discovery. Discovery-only providers (e.g. a models.yml provider with - * `discovery:` and no static models) can hold zero models until the online - * refresh completes, so a models-only check would misreport them as - * unknown during session construction. - */ - #isKnownProvider(provider: string): boolean { - return this.#modelRegistry.hasProvider(provider); - } - - #getRetryFallbackChains(): RetryFallbackChains { - const configuredChains = this.settings.get("retry.fallbackChains"); - if (!configuredChains || typeof configuredChains !== "object") return {}; - const chains: RetryFallbackChains = { ...(configuredChains as RetryFallbackChains) }; - const defaultChain = chains.default; - if (Array.isArray(defaultChain)) { - for (const role of Object.keys(this.settings.getModelRoles())) { - if (role !== "default" && chains[role] === undefined) { - chains[role] = defaultChain; - } - } - } - return chains; - } - - #validateRetryFallbackChains(): void { - const configuredChains = this.settings.get("retry.fallbackChains"); - if (configuredChains === undefined) return; - if (!configuredChains || typeof configuredChains !== "object" || Array.isArray(configuredChains)) { - const msg = "retry.fallbackChains must be a mapping of role names or model selectors to selector arrays."; - logger.warn(msg); - this.configWarnings.push(msg); - return; - } - - for (const key in configuredChains) { - const chain = (configuredChains as RetryFallbackChains)[key]; - const keyKind = isRetryFallbackModelKey(key) ? "model" : "role"; - if (keyKind === "model") { - if (isRetryFallbackWildcardKey(key)) { - const { provider } = parseRetryFallbackWildcard(key, p => this.#isKnownProvider(p)); - if (!this.#isKnownProvider(provider)) { - const msg = `retry.fallbackChains wildcard key references unknown provider: ${key}`; - logger.warn(msg); - this.configWarnings.push(msg); - } - } else { - const parsedKey = parseRetryFallbackSelector(key, this.#modelRegistry); - if (!parsedKey) { - const msg = `Invalid model selector key in retry.fallbackChains: ${key}`; - logger.warn(msg); - this.configWarnings.push(msg); - } else if (!this.#modelRegistry.find(parsedKey.provider, parsedKey.id)) { - const msg = `retry.fallbackChains key references unknown model: ${key}`; - logger.warn(msg); - this.configWarnings.push(msg); - } - } - } - if (!Array.isArray(chain)) { - const msg = `Fallback chain for ${keyKind} '${key}' must be an array of selector strings.`; - logger.warn(msg); - this.configWarnings.push(msg); - continue; - } - for (const selectorStr of chain) { - if (typeof selectorStr !== "string") { - const msg = `Fallback chain for ${keyKind} '${key}' contains a non-string selector.`; - logger.warn(msg); - this.configWarnings.push(msg); - continue; - } - if (isRetryFallbackWildcardKey(selectorStr)) { - const { provider } = parseRetryFallbackWildcard(selectorStr, p => this.#isKnownProvider(p)); - if (!this.#isKnownProvider(provider)) { - const msg = `Fallback chain for ${keyKind} '${key}' references unknown provider: ${selectorStr}`; - logger.warn(msg); - this.configWarnings.push(msg); - } - continue; - } - const parsed = parseRetryFallbackSelector(selectorStr, this.#modelRegistry); - if (!parsed) { - const msg = `Invalid fallback selector format in ${keyKind} '${key}': ${selectorStr}`; - logger.warn(msg); - this.configWarnings.push(msg); - continue; - } - const exists = this.#modelRegistry.find(parsed.provider, parsed.id); - if (!exists) { - const msg = `Fallback chain for ${keyKind} '${key}' references unknown model: ${selectorStr}`; - logger.warn(msg); - this.configWarnings.push(msg); - } - } - } - } - - #getRetryFallbackRevertPolicy(): RetryFallbackRevertPolicy { - return this.settings.get("retry.fallbackRevertPolicy") === "never" ? "never" : "cooldown-expiry"; - } - - #getRetryFallbackPrimarySelector(role: string): RetryFallbackSelector | undefined { - if (isRetryFallbackWildcardKey(role)) return undefined; - if (isRetryFallbackModelKey(role)) return parseRetryFallbackSelector(role, this.#modelRegistry); - const configuredSelector = this.settings.getModelRole(role); - return configuredSelector ? parseRetryFallbackSelector(configuredSelector, this.#modelRegistry) : undefined; - } - - #clearActiveRetryFallback(): void { - this.#activeRetryFallback = undefined; - } - - #isRetryFallbackSelectorSuppressed(selector: RetryFallbackSelector): boolean { - return this.#modelRegistry.isSelectorSuppressed(selector.raw); - } - - #noteRetryFallbackCooldown(currentSelector: string, retryAfterMs: number | undefined, errorMessage: string): void { - let cooldownMs = retryAfterMs; - if (!cooldownMs || cooldownMs <= 0) { - const reason = parseRateLimitReason(errorMessage); - cooldownMs = reason === "UNKNOWN" ? 5 * 60 * 1000 : calculateRateLimitBackoffMs(reason); - } - this.#modelRegistry.suppressSelector(currentSelector, Date.now() + cooldownMs); - } - - /** - * Map the failing model selector to the chain key that owns it, by - * specificity: an exact model-selector key, then a `provider/*` wildcard, - * then a model role whose current assignment matches, then `default`. - * Model-oriented keys win over roles so a chain follows the model across - * role reassignments. - */ - #resolveRetryFallbackRole( - currentSelector: string, - currentModel: Model | null | undefined = this.model, - ): string | undefined { - const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); - if (!parsedCurrent) return undefined; - const chains = this.#getRetryFallbackChains(); - const currentBaseSelector = formatRetryFallbackBaseSelector(parsedCurrent); - const currentPlainSelector = currentModel - ? formatModelSelectorValue(formatModelString(currentModel), parsedCurrent.thinkingLevel) - : undefined; - const currentPlainBaseSelector = - currentPlainSelector && currentPlainSelector !== currentSelector - ? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent) - : undefined; - - const exactModelKeys: string[] = []; - const roleKeys: string[] = []; - for (const key in chains) { - if (!isRetryFallbackModelKey(key)) roleKeys.push(key); - else if (!isRetryFallbackWildcardKey(key)) exactModelKeys.push(key); - } - const matchesCurrent = (primary: RetryFallbackSelector | undefined): boolean => { - if (!primary) return false; - if (primary.raw === currentSelector || (currentPlainSelector && primary.raw === currentPlainSelector)) { - return true; - } - const base = formatRetryFallbackBaseSelector(primary); - return base === currentBaseSelector || (!!currentPlainBaseSelector && base === currentPlainBaseSelector); - }; - - // 1. Exact model-selector keys — most specific. - for (const key of exactModelKeys) { - if (matchesCurrent(this.#getRetryFallbackPrimarySelector(key))) return key; - } - // 2. Provider wildcards — an id-prefixed key (`openrouter/google/*`) - // beats the plain `provider/*` key for ids under its prefix. - let wildcardMatch: string | undefined; - let wildcardPrefixLength = -1; - for (const key in chains) { - if (!isRetryFallbackWildcardKey(key) || !Array.isArray(chains[key])) continue; - const { provider, idPrefix } = parseRetryFallbackWildcard(key, p => this.#isKnownProvider(p)); - if (provider !== parsedCurrent.provider) continue; - if (idPrefix !== undefined && !parsedCurrent.id.startsWith(`${idPrefix}/`)) continue; - const prefixLength = idPrefix === undefined ? 0 : idPrefix.length; - if (prefixLength > wildcardPrefixLength) { - wildcardMatch = key; - wildcardPrefixLength = prefixLength; - } - } - if (wildcardMatch) return wildcardMatch; - // 3. Role keys — matched by the role's currently-assigned model. - for (const key of roleKeys) { - if (matchesCurrent(this.#getRetryFallbackPrimarySelector(key))) return key; - } - // 4. The default chain, when default has no explicit role primary. - const defaultChain = chains.default; - if ( - Array.isArray(defaultChain) && - defaultChain.length > 0 && - this.#getRetryFallbackPrimarySelector("default") === undefined - ) { - return "default"; - } - return undefined; - } - - /** - * Parse one configured chain entry. A `provider/*` entry keeps the failing - * model's id and swaps the provider (google-antigravity/x → google/x); an - * id-prefixed `provider/prefix/*` entry re-prefixes the failing model's - * bare id instead (openrouter/google/* : google-antigravity/x → - * openrouter/google/x). Ids the target provider lacks are skipped by the - * candidate loop's registry lookup. - */ - #parseRetryFallbackChainEntry( - entry: string, - current: RetryFallbackSelector | undefined, - ): RetryFallbackSelector | undefined { - if (isRetryFallbackWildcardKey(entry)) { - if (!current) return undefined; - const { provider, idPrefix } = parseRetryFallbackWildcard(entry, p => this.#isKnownProvider(p)); - const bareId = current.id.slice(current.id.lastIndexOf("/") + 1); - let id: string; - if (idPrefix !== undefined) { - id = `${idPrefix}/${bareId}`; - } else if ( - bareId !== current.id && - !this.#modelRegistry.find(provider, current.id) && - this.#modelRegistry.find(provider, bareId) - ) { - // Aggregator → direct: the failing id carries a vendor prefix the - // target provider does not use (openrouter/google/x → google-vertex/x). - id = bareId; - } else { - id = current.id; - } - return { raw: `${provider}/${id}`, provider, id, thinkingLevel: undefined }; - } - return parseRetryFallbackSelector(entry, this.#modelRegistry); - } - - #getRetryFallbackEffectiveChain(role: string, currentSelector?: string): RetryFallbackSelector[] { - const parsedCurrent = currentSelector - ? parseRetryFallbackSelector(currentSelector, this.#modelRegistry) - : undefined; - const seen = new Set(); - const chain: RetryFallbackSelector[] = []; - if (isRetryFallbackWildcardKey(role)) { - // A wildcard key has no fixed primary: the active model is the - // primary, followed by the configured provider-level fallbacks. - if (parsedCurrent) { - chain.push(parsedCurrent); - seen.add(parsedCurrent.raw); - } - } else { - const primarySelector = this.#getRetryFallbackPrimarySelector(role); - if (!primarySelector) return []; - chain.push(primarySelector); - seen.add(primarySelector.raw); - } - for (const selector of this.#getRetryFallbackChains()[role] ?? []) { - const parsed = this.#parseRetryFallbackChainEntry(selector, parsedCurrent); - if (!parsed || seen.has(parsed.raw)) continue; - seen.add(parsed.raw); - chain.push(parsed); - } - return chain; - } - - #findRetryFallbackCandidates( - role: string, - currentSelector: string, - currentModel: Model | null | undefined = this.model, - ): RetryFallbackSelector[] { - let chain = this.#getRetryFallbackEffectiveChain(role, currentSelector); - const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); - if (chain.length === 0 && role === "default" && parsedCurrent) { - const chains = this.#getRetryFallbackChains(); - const defaultChain = chains.default; - if ( - Array.isArray(defaultChain) && - defaultChain.length > 0 && - this.#getRetryFallbackPrimarySelector("default") === undefined - ) { - const seen = new Set([parsedCurrent.raw]); - chain = [parsedCurrent]; - for (const selector of defaultChain) { - const parsed = this.#parseRetryFallbackChainEntry(selector, parsedCurrent); - if (!parsed || seen.has(parsed.raw)) continue; - seen.add(parsed.raw); - chain.push(parsed); - } - } - } - if (chain.length <= 1) return []; - const currentBaseSelector = parsedCurrent ? formatRetryFallbackBaseSelector(parsedCurrent) : undefined; - const currentPlainSelector = - currentModel && parsedCurrent - ? formatModelSelectorValue(formatModelString(currentModel), parsedCurrent.thinkingLevel) - : undefined; - const currentPlainBaseSelector = - parsedCurrent && currentPlainSelector && currentPlainSelector !== currentSelector - ? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent) - : undefined; - const exactIndex = chain.findIndex( - selector => selector.raw === currentSelector || selector.raw === currentPlainSelector, - ); - if (exactIndex >= 0) return chain.slice(exactIndex + 1); - const baseIndex = currentBaseSelector - ? chain.findIndex(selector => { - const selectorBase = formatRetryFallbackBaseSelector(selector); - return selectorBase === currentBaseSelector || selectorBase === currentPlainBaseSelector; - }) - : -1; - if (baseIndex >= 0) return chain.slice(baseIndex + 1); - return chain.slice(1); - } - - async #applyRetryFallbackCandidate( - role: string, - selector: RetryFallbackSelector, - currentSelector: string, - options?: { pinFallback?: boolean }, - ): Promise { - const resolved = resolveModelOverride([selector.raw], this.#modelRegistry, this.settings); - const candidate = resolved.model ?? this.#modelRegistry.find(selector.provider, selector.id); - if (!candidate) { - throw new Error(`Retry fallback model not found: ${selector.raw}`); - } - const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId); - if (!apiKey) { - throw new Error(`No API key for retry fallback ${selector.raw}`); - } - - // Capture the configured selector (auto-aware) so a fallback chain preserves - // `auto` instead of collapsing it to the level it resolved to this turn. - const currentThinkingLevel = this.configuredThinkingLevel(); - const nextThinkingLevel = selector.thinkingLevel ?? currentThinkingLevel; - const candidateSelector = formatModelStringWithRouting(candidate); - this.#setModelWithProviderSessionReset(candidate); - this.sessionManager.appendModelChange(candidateSelector, EPHEMERAL_MODEL_CHANGE_ROLE); - this.settings.getStorage()?.recordModelUsage(candidateSelector); - this.setThinkingLevel(nextThinkingLevel); - if (!this.#activeRetryFallback) { - this.#activeRetryFallback = { - role, - originalSelector: currentSelector, - originalThinkingLevel: currentThinkingLevel, - lastAppliedFallbackThinkingLevel: nextThinkingLevel, - pinned: options?.pinFallback === true, - }; - } else { - this.#activeRetryFallback.lastAppliedFallbackThinkingLevel = nextThinkingLevel; - this.#activeRetryFallback.pinned = this.#activeRetryFallback.pinned || options?.pinFallback === true; - } - await this.#emitSessionEvent({ - type: "retry_fallback_applied", - from: currentSelector, - to: selector.raw, - role, - }); - } - - async #tryRetryModelFallback(currentSelector: string, options?: { pinFallback?: boolean }): Promise { - const role = this.#activeRetryFallback?.role ?? this.#resolveRetryFallbackRole(currentSelector); - if (!role) return false; - - for (const selector of this.#findRetryFallbackCandidates(role, currentSelector)) { - if (this.#isRetryFallbackSelectorSuppressed(selector)) continue; - const resolved = resolveModelOverride([selector.raw], this.#modelRegistry, this.settings); - const candidate = resolved.model ?? this.#modelRegistry.find(selector.provider, selector.id); - if (!candidate) continue; - const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId); - if (!apiKey) continue; - await this.#applyRetryFallbackCandidate(role, selector, currentSelector, options); - return true; - } - - return false; - } - - /** The active model when it is a Fireworks Fast (`-fast`) variant, else undefined. */ - #activeFireworksFastModel(): Model | undefined { - const model = this.model; - return model?.provider === "fireworks" && isFireworksFastModelId(model.id) ? model : undefined; - } - - /** - * True when the current turn failed on a Fireworks Fast (`-fast`) model in a - * way that should degrade to the reliable base (Standard) model. Fast is a - * speed-optimized router with no SLA, so any *pre-content* failure — a - * transient overload/5xx or a hard "router/model not found / unsupported" — - * is worth retrying on the base id. Skips failures the base model shares: - * context overflow (compaction's job), usage limits and auth errors (same - * account/key), and turns that already emitted a tool call (replaying would - * duplicate work). Requires the base model to exist in the registry. - */ - #isFireworksFastFallbackEligible(message: AssistantMessage): boolean { - const model = this.#activeFireworksFastModel(); - if (!model) return false; - if (message.stopReason !== "error") return false; - if (message.content.some(block => block.type === "toolCall")) return false; - // A content refusal/sensitivity stop is the model's decision, not a route - // failure — switching to the base model would just re-trigger it. - if (this.#isClassifierRefusal(message)) return false; - const id = this.#classifyRetryMessage(message); - if (AIError.isContextOverflow(message, model.contextWindow ?? 0)) return false; - if (AIError.is(id, AIError.Flag.UsageLimit)) return false; - if (AIError.is(id, AIError.Flag.AuthFailed)) return false; - return this.#modelRegistry.find("fireworks", toFireworksBaseModelId(model.id)) !== undefined; - } - - /** - * True when a turn failed with a hard (non-retryable) provider error but a - * configured `retry.fallbackChains` entry covers the active model: the same - * model is not worth retrying, yet a DIFFERENT model is a fresh chance, so - * the chain is consulted before the error becomes final. Skips failures a - * model switch cannot fix or must not replay: cancellations (abort-flavored - * errors are not model faults), context overflow (compaction's job), - * classifier refusals (chain consult is handled on the retryable path with - * `pinFallback`), and turns that already emitted a tool call (replaying - * could duplicate work). - */ - #isHardErrorFallbackEligible(message: AssistantMessage): boolean { - if (message.stopReason !== "error") return false; - const model = this.model; - if (!model) return false; - const retrySettings = this.settings.getGroup("retry"); - if (!retrySettings.enabled || !retrySettings.modelFallback) return false; - if (this.#isClassifierRefusal(message)) return false; - const id = this.#classifyRetryMessage(message); - if (AIError.is(id, AIError.Flag.Abort) || AIError.is(id, AIError.Flag.UserInterrupt)) return false; - if (AIError.isContextOverflow(message, model.contextWindow ?? 0)) return false; - if (this.#hasReplayUnsafeToolOutput(message)) return false; - const currentSelector = formatRetryFallbackSelector(model, this.thinkingLevel); - const role = this.#activeRetryFallback?.role ?? this.#resolveRetryFallbackRole(currentSelector); - if (!role) return false; - return this.#findRetryFallbackCandidates(role, currentSelector).length > 0; - } - - /** - * Switch the active model from a Fireworks Fast (`-fast`) variant to its base - * (Standard) id and stick there for the rest of the session — the auto - * fallback that makes Fast a safe default. Returns false when the current - * model is not a fast variant, the base id is missing, or it has no key. - */ - async #tryFireworksFastFallback(currentSelector: string): Promise { - const model = this.#activeFireworksFastModel(); - if (!model) return false; - const baseModel = this.#modelRegistry.find("fireworks", toFireworksBaseModelId(model.id)); - if (!baseModel) return false; - const apiKey = await this.#modelRegistry.getApiKey(baseModel, this.sessionId); - if (!apiKey) return false; - const baseSelector = formatModelStringWithRouting(baseModel); - this.#setModelWithProviderSessionReset(baseModel); - this.sessionManager.appendModelChange(baseSelector, EPHEMERAL_MODEL_CHANGE_ROLE); - this.settings.getStorage()?.recordModelUsage(baseSelector); - await this.#emitSessionEvent({ - type: "retry_fallback_applied", - from: currentSelector, - to: baseSelector, - role: "fireworks-fast", - }); - return true; - } - - async #maybeRestoreRetryFallbackPrimary(): Promise { - if (!this.#activeRetryFallback) return; - if (this.#activeRetryFallback.pinned) return; - if (this.#getRetryFallbackRevertPolicy() !== "cooldown-expiry") return; - - const { - originalSelector: originalSelectorRaw, - originalThinkingLevel, - lastAppliedFallbackThinkingLevel, - } = this.#activeRetryFallback; - const originalSelector = parseRetryFallbackSelector(originalSelectorRaw, this.#modelRegistry); - if (!originalSelector) { - this.#clearActiveRetryFallback(); - return; - } - - const currentModel = this.model; - if (!currentModel) return; - const currentSelector = formatRetryFallbackSelector(currentModel, this.thinkingLevel); - if (currentSelector === originalSelector.raw) { - if (!this.#isRetryFallbackSelectorSuppressed(originalSelector)) { - this.#clearActiveRetryFallback(); - } - return; - } - if (this.#isRetryFallbackSelectorSuppressed(originalSelector)) return; - - const resolvedPrimary = resolveModelOverride([originalSelector.raw], this.#modelRegistry, this.settings); - const primaryModel = - resolvedPrimary.model ?? this.#modelRegistry.find(originalSelector.provider, originalSelector.id); - if (!primaryModel) return; - const apiKey = await this.#modelRegistry.getApiKey(primaryModel, this.sessionId); - if (!apiKey) return; - - const currentThinkingLevel = this.configuredThinkingLevel(); - const thinkingToApply = - currentThinkingLevel === lastAppliedFallbackThinkingLevel ? originalThinkingLevel : currentThinkingLevel; - const primarySelector = formatModelStringWithRouting(primaryModel); - this.#setModelWithProviderSessionReset(primaryModel); - this.sessionManager.appendModelChange(primarySelector, EPHEMERAL_MODEL_CHANGE_ROLE); - this.settings.getStorage()?.recordModelUsage(primarySelector); - this.setThinkingLevel(thinkingToApply); - this.#clearActiveRetryFallback(); - } - - #parseRetryAfterMsFromError(errorMessage: string): number | undefined { - const now = Date.now(); - const retryAfterMsMatch = /retry-after-ms\s*[:=]\s*(\d+)/i.exec(errorMessage); - if (retryAfterMsMatch) { - return Math.max(0, Number(retryAfterMsMatch[1])); - } - - const retryAfterMatch = /retry-after\s*[:=]\s*([^\s,;]+)/i.exec(errorMessage); - if (retryAfterMatch) { - const value = retryAfterMatch[1]; - const seconds = Number(value); - if (!Number.isNaN(seconds)) { - return Math.max(0, seconds * 1000); - } - const dateMs = Date.parse(value); - if (!Number.isNaN(dateMs)) { - return Math.max(0, dateMs - now); - } - } - - const retryHintMs = extractRetryHint(undefined, errorMessage); - if (retryHintMs !== undefined) { - return retryHintMs; - } - - const resetMsMatch = /x-ratelimit-reset-ms\s*[:=]\s*(\d+)/i.exec(errorMessage); - if (resetMsMatch) { - const resetMs = Number(resetMsMatch[1]); - if (!Number.isNaN(resetMs)) { - if (resetMs > 1_000_000_000_000) { - return Math.max(0, resetMs - now); - } - return Math.max(0, resetMs); - } - } - - const resetMatch = /x-ratelimit-reset\s*[:=]\s*(\d+)/i.exec(errorMessage); - if (resetMatch) { - const resetSeconds = Number(resetMatch[1]); - if (!Number.isNaN(resetSeconds)) { - if (resetSeconds > 1_000_000_000) { - return Math.max(0, resetSeconds * 1000 - now); - } - return Math.max(0, resetSeconds * 1000); - } - } - - // Smart Fallback if no exact headers found - return undefined; - } - - /** - * Handle retryable errors with exponential backoff, credential rotation, and - * model-fallback chains. Also entered for NON-retryable errors when a switch - * is the recovery (`fireworksFastFallback`, `hardErrorFallback`): then a - * successful model switch retries immediately, and a failed switch surfaces - * the error without a same-model backoff retry. - * @returns true if retry was initiated, false if max retries exceeded or disabled - */ - async #handleRetryableError( - message: AssistantMessage, - options?: { - allowModelFallback?: boolean; - fireworksFastFallback?: boolean; - hardErrorFallback?: boolean; - preserveFailedTurn?: boolean; - }, - ): Promise { - const retrySettings = this.settings.getGroup("retry"); - // The Fireworks Fast→base degrade is an intrinsic model-selection safety net, - // not a retry loop, so it runs even when the user disabled retries: it switches - // the model once and lets the base turn proceed. - if (!retrySettings.enabled && !options?.fireworksFastFallback) return false; - const classifierRefusal = this.#isClassifierRefusal(message); - - const generation = this.#promptGeneration; - this.#retryAttempt++; - - // Create retry promise on first attempt so waitForRetry() can await it - // Ensure only one promise exists (avoid orphaned promises from concurrent calls) - if (!this.#retryPromise) { - const { promise, resolve } = Promise.withResolvers(); - this.#retryPromise = promise; - this.#retryResolve = resolve; - } - - // All attempts on the current model are spent. Don't fail yet: the - // fallback chain below gets one last consult. Credential rotation can - // consume the entire budget without the fallback branch ever running - // (every rotation sets switchedCredential and skips it), so without - // this last resort a provider-wide usage cap never fails over to the - // configured chain. - const maxRetries = this.#isOpenRouterThinkingStreamClose(message) - ? Math.min(retrySettings.maxRetries, 1) - : retrySettings.maxRetries; - const retryBudgetExhausted = this.#retryAttempt > maxRetries; - - const errorMessage = message.errorMessage || "Unknown error"; - const id = this.#classifyRetryMessage(message); - const staleOpenAIResponsesReplayError = AIError.is(id, AIError.Flag.StaleResponsesItem); - const parsedRetryAfterMs = this.#parseRetryAfterMsFromError(errorMessage); - let delayMs = staleOpenAIResponsesReplayError - ? 0 - : calculateRetryBackoffDelayMs(retrySettings.baseDelayMs, this.#retryAttempt); - let switchedCredential = false; - let switchedModel = false; - // Set when a usage-limit error pinned the wait to credential - // availability — suppresses the generic retry-after bump below. - let usageLimitWaitMs: number | undefined; - - if (staleOpenAIResponsesReplayError) { - this.#resetCurrentResponsesProviderSession("stale replay error"); - } - - if ( - !retryBudgetExhausted && - this.model && - !staleOpenAIResponsesReplayError && - AIError.is(id, AIError.Flag.UsageLimit) - ) { - const retryAfterMs = parsedRetryAfterMs ?? calculateRateLimitBackoffMs(parseRateLimitReason(errorMessage)); - const outcome = await this.#modelRegistry.authStorage.markUsageLimitReached( - this.model.provider, - this.sessionId, - { - retryAfterMs, - baseUrl: this.model.baseUrl, - modelId: this.model.id, - }, - ); - if (outcome.switched) { - switchedCredential = true; - delayMs = 0; - } else if (await this.#maybeAutoRedeemCodexReset()) { - // A live usage-limit 429 on the active Codex account, with a banked - // reset and the opt-in setting on: spend the reset and retry - // immediately instead of waiting out the window. Runs after the - // free sibling-switch above and before model fallback below. - switchedCredential = true; - delayMs = 0; - } else { - // No sibling credential is usable right now. Wait for whichever - // comes first: the provider's retry-after window for the current - // account, or the earliest moment a temporarily blocked sibling - // frees up (e.g. a 60s post-401 block or a 5-min usage-probe - // block) — the next attempt's getApiKey re-ranks and picks it up. - // Without this, one short-lived sibling block escalates a - // recoverable situation into the provider's multi-hour wait and - // trips the fail-fast cap below. - usageLimitWaitMs = retryAfterMs; - if (outcome.retryAtMs !== undefined) { - const siblingWaitMs = Math.max(0, outcome.retryAtMs - Date.now()) + SIBLING_UNBLOCK_BUFFER_MS; - if (siblingWaitMs < usageLimitWaitMs) { - usageLimitWaitMs = siblingWaitMs; - } - } - if (usageLimitWaitMs > delayMs) { - delayMs = usageLimitWaitMs; - } - } - } - - const allowModelFallback = options?.allowModelFallback !== false; - const currentSelector = this.model ? formatRetryFallbackSelector(this.model, this.thinkingLevel) : undefined; - if (!staleOpenAIResponsesReplayError && !switchedCredential && currentSelector) { - // A refusal chain stops at the retry budget: the exhausted-attempt - // last resort is for provider failures, not classifier decisions. - if (allowModelFallback && retrySettings.modelFallback && !(retryBudgetExhausted && classifierRefusal)) { - if (!classifierRefusal) { - this.#noteRetryFallbackCooldown(currentSelector, parsedRetryAfterMs, errorMessage); - } - switchedModel = await this.#tryRetryModelFallback(currentSelector, { pinFallback: classifierRefusal }); - } - // Auto fallback from a Fireworks Fast variant to its base model. Independent - // of the role-fallback setting: it's intrinsic to the Fast contract (speed - // best-effort, degrade to Standard on failure) and triggers on hard router - // errors the generic retry classifier would otherwise reject. - if (!switchedModel && allowModelFallback && options?.fireworksFastFallback) { - switchedModel = await this.#tryFireworksFastFallback(currentSelector); - } - if (switchedModel) { - delayMs = 0; - } else if (usageLimitWaitMs === undefined && parsedRetryAfterMs && parsedRetryAfterMs > delayMs) { - delayMs = parsedRetryAfterMs; - } - } - if (retryBudgetExhausted) { - if (!switchedModel) { - await this.#persistTerminalEmptyErrorTurn(message); - // Max retries exceeded and no fallback model to switch to: emit - // final failure and reset. - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt: this.#retryAttempt - 1, - finalError: message.errorMessage, - }); - this.#clearPendingRecoveredRetryErrors(); - this.#retryAttempt = 0; - this.#resolveRetry(); // Resolve so waitForRetry() completes - return false; - } - // The fallback model gets a fresh retry budget — leaving the spent - // counter in place would exhaust it again on its first error. - this.#retryAttempt = 1; - } - if (classifierRefusal && !switchedModel) { - // A prior attempt in this saga already announced `auto_retry_start` - // (retryAttempt was incremented for each call to this method, so > 1 - // means at least one earlier attempt started the loop) but this - // attempt is not going to retry — the saga must close with its own - // `auto_retry_end` so subscribers tracking retry-outstanding state - // (e.g. suppressing a duplicate error toast) don't stay latched on - // an announcement that never resolves. - if (this.#retryAttempt > 1) { - await this.#persistTerminalEmptyErrorTurn(message); - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt: this.#retryAttempt - 1, - finalError: errorMessage, - }); - this.#clearPendingRecoveredRetryErrors(); - } - this.#retryAttempt = 0; - this.#resolveRetry(); - return false; - } - // A fallback switch was the whole reason we entered (Fast→base degrade or - // a hard-error chain consult) but it could not happen (e.g. no candidate - // has a credential). Don't fall through to backing-off and retrying the - // failing model for an error the generic classifier wouldn't retry — - // surface it instead. - if ( - (options?.fireworksFastFallback || options?.hardErrorFallback) && - !switchedModel && - !this.#isRetryableError(message) - ) { - // Same auto_retry_end backstop as the classifier-refusal branch above. - if (this.#retryAttempt > 1) { - await this.#persistTerminalEmptyErrorTurn(message); - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt: this.#retryAttempt - 1, - finalError: errorMessage, - }); - this.#clearPendingRecoveredRetryErrors(); - } - this.#retryAttempt = 0; - this.#resolveRetry(); - return false; - } - - // Fail-fast cap: if the provider asks us to wait longer than - // retry.maxDelayMs and we have no fallback credential or model to - // switch to, surface the error instead of sleeping. Defends against - // 3-hour Anthropic rate-limit windows that would otherwise leave a - // subagent (or interactive session) silently hung. The original - // assistant error message is preserved in agent state so the caller - // can act on it. - const maxDelayMs = retrySettings.maxDelayMs; - if (maxDelayMs > 0 && delayMs > maxDelayMs && !switchedCredential && !switchedModel) { - await this.#persistTerminalEmptyErrorTurn(message); - const attempt = this.#retryAttempt; - this.#retryAttempt = 0; - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt, - finalError: `Provider requested ${delayMs}ms wait, exceeds retry.maxDelayMs (${maxDelayMs}ms). Original error: ${errorMessage}`, - }); - this.#clearPendingRecoveredRetryErrors(); - this.#resolveRetry(); - return false; - } - - await this.#recordPendingRecoveredRetryError(message, id, { switchedCredential, switchedModel, delayMs }); - - await this.#emitSessionEvent({ - type: "auto_retry_start", - attempt: this.#retryAttempt, - maxAttempts: maxRetries, - delayMs, - errorMessage, - errorId: message.errorId, - }); - - // Resolved stream-stall tools have already emitted results. Keep that failed - // turn intact so continuation cannot repeat their side effects. - if (!options?.preserveFailedTurn) { - this.#removeAssistantMessageFromActiveContext(message, "auto-retry"); - } - - // A thinking/response loop retried into identical context loops again. Inject a - // hidden redirect so the retried turn sees a directive to break the repeated - // pattern instead of re-sampling the same stalled reasoning. - this.#maybeInjectThinkingLoopRedirect(id); - - // Wait with exponential backoff (abortable). - const retryAbortController = new AbortController(); - this.#retryAbortController?.abort(); - this.#retryAbortController = retryAbortController; - try { - await scheduler.wait(delayMs, { signal: retryAbortController.signal }); - } catch { - if (this.#retryAbortController !== retryAbortController) { - return false; - } - // Aborted during sleep - emit end event so UI can clean up - const attempt = this.#retryAttempt; - this.#retryAttempt = 0; - this.#retryAbortController = undefined; - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt, - finalError: "Retry cancelled", - }); - this.#clearPendingRecoveredRetryErrors(); - this.#resolveRetry(); - return false; - } - if (this.#retryAbortController === retryAbortController) { - this.#retryAbortController = undefined; - } - - // Retry via continue() outside the agent_end event callback chain. - this.#scheduleAgentContinue({ delayMs: 1, generation }); - - return true; - } - - /** - * Inject a hidden redirect notice when a thinking/response loop is being retried, so - * the retried turn carries an instruction to break the repeated pattern instead of - * re-sampling the same stalled context. Injected on every {@link AIError.Flag.ThinkingLoop} - * retry (the failed assistant is dropped each attempt, so the notice does not accumulate - * unboundedly). No-op unless `id` carries the ThinkingLoop flag and the loop guard is - * enabled. The notice is generic on purpose — the detector's detail can quote raw model - * text, which must not be interpolated into a higher-priority developer message. - */ - #maybeInjectThinkingLoopRedirect(id: number): void { - if (!AIError.is(id, AIError.Flag.ThinkingLoop)) return; - if (this.settings.get("model.loopGuard.enabled") !== true) return; - this.agent.appendMessage({ - role: "custom", - customType: THINKING_LOOP_REDIRECT_TYPE, - content: thinkingLoopRedirectTemplate, - display: false, - attribution: "agent", - timestamp: Date.now(), - }); - this.sessionManager.appendCustomMessageEntry( - THINKING_LOOP_REDIRECT_TYPE, - thinkingLoopRedirectTemplate, - false, - undefined, - "agent", - ); - } - - /** - * Cancel in-progress retry. - */ + /** Cancel an in-progress retry. */ abortRetry(): void { - this.#retryAbortController?.abort(); - // Note: _retryAttempt is reset in the catch block of _autoRetry - this.#resolveRetry(); + this.#recovery.abortRetry(); } - async #promptAgentWithIdleRetry(messages: AgentMessage[], options?: { toolChoice?: ToolChoice }): Promise { - const deadline = Date.now() + 30_000; - for (;;) { - try { - await this.agent.prompt(messages, options); - return; - } catch (err) { - if (!(err instanceof AgentBusyError)) { - throw err; - } - if (Date.now() >= deadline) { - throw new Error("Timed out waiting for prior agent run to finish before prompting."); - } - await this.agent.waitForIdle(); - } - } - } - - /** Whether auto-retry is currently in progress */ + /** Whether auto-retry is currently in progress. */ get isRetrying(): boolean { - return this.#retryPromise !== undefined; + return this.#recovery.isRetrying; } - /** Whether auto-retry is enabled */ + /** Whether auto-retry is enabled. */ get autoRetryEnabled(): boolean { - return this.settings.get("retry.enabled") ?? true; + return this.#recovery.autoRetryEnabled; } - /** - * Toggle auto-retry setting. - */ + /** Toggle the auto-retry setting. */ setAutoRetryEnabled(enabled: boolean): void { - this.settings.set("retry.enabled", enabled); + this.#recovery.setAutoRetryEnabled(enabled); } - /** - * Manually retry the last failed assistant turn. - * Removes the error message from active agent state when present and - * re-attempts with a fresh retry budget. - * - * A stream that stalls or aborts mid-tool-call ends the turn with - * `stopReason: "error" | "aborted"` and then appends one synthetic - * {@link isSyntheticToolResultMessage tool_result} per emitted tool call to - * preserve the provider's tool_use/tool_result pairing (see - * `createAbortedToolResult` in `agent-loop.ts`). Those placeholders trail the - * failed assistant turn, so the retry lookback walks back over them before - * checking the assistant message; it strips both the placeholders and the - * failed turn before re-attempting. - * - * A restored session deliberately omits failed assistant turns from provider - * context. In that case, the persisted display transcript remains the source - * of truth for whether the current branch has a retryable failed tail. - * - * @returns true if retry was initiated, false if no failed turn to retry or agent is busy - */ - async retry(): Promise { - if (this.isStreaming || this.isCompacting || this.isRetrying) return false; - const messages = this.agent.state.messages; - const activeTurnEnd = retryableAssistantTurnEnd(messages); - if (activeTurnEnd !== undefined) { - // Remove the failed/aborted assistant message plus its synthetic tool - // results (same as auto-retry does before re-attempting). - this.agent.replaceMessages(messages.slice(0, activeTurnEnd - 1)); - } else { - // A restored session already dropped the failed assistant turn (and its - // paired synthetic tool results) from provider context, so the persisted - // display transcript is the source of truth for a retryable failed tail. - const transcriptMessages = this.sessionManager.buildSessionContext({ transcript: true }).messages; - if (retryableAssistantTurnEnd(transcriptMessages) === undefined) return false; - } - - // Reset retry budget for a fresh attempt - this.#retryAttempt = 0; - - // Re-attempt the turn - this.#scheduleAgentContinue({ delayMs: 1 }); - - return true; + /** Retry the last failed assistant turn when the session is idle. */ + retry(): Promise { + return this.#recovery.retry(); } // ========================================================================= // Bash Execution // ========================================================================= - async #saveBashOriginalArtifact(target: BashSessionTarget, originalText: string): Promise { - try { - const destination = target.destination ?? (await target.pending); - return await destination?.manager.saveArtifact(originalText, "bash-original"); - } catch { - return undefined; - } - } - - #createBashMessage( - command: string, - result: BashResult, - options?: { excludeFromContext?: boolean }, - ): BashExecutionMessage { - const meta = outputMeta().truncationFromSummary(result, { direction: "tail" }).get(); - return { - role: "bashExecution", - command, - output: result.output, - exitCode: result.exitCode, - cancelled: result.cancelled, - truncated: result.truncated, - meta, - timestamp: Date.now(), - excludeFromContext: options?.excludeFromContext, - }; - } - - #captureBashSessionTarget(): BashSessionTarget { - this.#bashSessionTarget.refs++; - return this.#bashSessionTarget; - } - - async #releaseBashSessionTarget(target: BashSessionTarget): Promise { - if (target.refs <= 0) throw new Error("Bash session target released more than once"); - target.refs--; - if (target.refs === 0 && target.destination?.kind === "detached") { - await target.destination.manager.close(); - } - } - - #appendBashMessage(destination: BashAppendDestination, message: BashExecutionMessage): void { - switch (destination.kind) { - case "current": - this.agent.appendMessage(message); - destination.manager.appendMessage(message); - break; - case "detached": - destination.manager.appendMessage(message); - break; - case "branch": - destination.parentId = destination.manager.appendMessageToBranch(message, destination.parentId); - break; - } - } - - async #appendOwnedBashMessage(target: BashSessionTarget, message: BashExecutionMessage): Promise { - try { - const destination = target.destination ?? (await target.pending); - if (!destination) throw new Error("Bash session target has no append destination"); - this.#appendBashMessage(destination, message); - } finally { - await this.#releaseBashSessionTarget(target); - } - } - - async #recordBashResultForTarget( - target: BashSessionTarget, - command: string, - result: BashResult, - options?: { excludeFromContext?: boolean }, - ): Promise { - const message = this.#createBashMessage(command, result, options); - if (this.isStreaming && target === this.#bashSessionTarget) { - this.#pendingBashMessages.push({ target, message }); - return; - } - await this.#appendOwnedBashMessage(target, message); - } - - /** Run a leaf rewrite while retaining any in-flight bash on its originating branch. */ - #withBashBranchTransition(mutate: () => T): T { - const bashTransition = this.#beginBashSessionTransition(); - let branchTransitioned = false; - try { - const result = mutate(); - this.#markBashSessionTransition(bashTransition); - branchTransitioned = true; - return result; - } finally { - this.#finishBashSessionTransition(bashTransition, branchTransitioned); - } - } - - /** - * Snapshot the session/branch that owns any in-flight bash before a transition. - * When an owner is still active, its target is detached to a clone so a failed - * or intentionally dropped transition never redirects the late result. - */ - #beginBashSessionTransition(options?: { persistDetached?: boolean }): BashSessionTransition { - const oldTarget = this.#bashSessionTarget; - let detachedManager: SessionManager | undefined; - let resolveOld: ((destination: BashAppendDestination) => void) | undefined; - if (oldTarget.refs > 0) { - detachedManager = this.sessionManager.cloneCurrentSession({ persist: options?.persistDetached }); - const pendingOld = Promise.withResolvers(); - oldTarget.destination = undefined; - oldTarget.pending = pendingOld.promise; - resolveOld = pendingOld.resolve; - } - - const pendingNew = Promise.withResolvers(); - return { - oldTarget, - newTarget: { - sessionId: this.sessionManager.getSessionId(), - refs: 0, - pending: pendingNew.promise, - }, - oldSessionId: this.sessionManager.getSessionId(), - oldSessionFile: this.sessionManager.getSessionFile(), - oldLeafId: this.sessionManager.getLeafId(), - detachedManager, - resolveOld, - resolveNew: pendingNew.resolve, - }; - } - - /** Adopt the transition's new target as the live bash owner. */ - #markBashSessionTransition(transition: BashSessionTransition): void { - transition.newTarget.sessionId = this.sessionManager.getSessionId(); - this.#bashSessionTarget = transition.newTarget; - } - - /** - * Resolve the pending append destinations opened by {@link #beginBashSessionTransition}. - * On success the old owner keeps its original session/branch (same file → current or - * branch destination; different file → detached clone); on failure both fall back to - * the still-current manager and the clone is discarded. - */ - #finishBashSessionTransition(transition: BashSessionTransition, success: boolean): void { - const currentDestination: BashAppendDestination = { kind: "current", manager: this.sessionManager }; - let oldDestination: BashAppendDestination = currentDestination; - if (success && transition.resolveOld) { - const currentFile = this.sessionManager.getSessionFile(); - const sameFile = - transition.oldSessionFile === currentFile || - (transition.oldSessionFile !== undefined && - currentFile !== undefined && - path.resolve(transition.oldSessionFile) === path.resolve(currentFile)); - const sameSession = transition.oldSessionId === this.sessionManager.getSessionId() && sameFile; - if (sameSession) { - oldDestination = - transition.oldLeafId === this.sessionManager.getLeafId() - ? currentDestination - : { kind: "branch", manager: this.sessionManager, parentId: transition.oldLeafId }; - } else if (transition.detachedManager) { - oldDestination = { kind: "detached", manager: transition.detachedManager }; - } - } - - if (transition.resolveOld) { - transition.oldTarget.pending = undefined; - transition.oldTarget.destination = oldDestination; - transition.resolveOld(oldDestination); - } - - transition.newTarget.pending = undefined; - transition.newTarget.destination = currentDestination; - if (!success) transition.newTarget.sessionId = this.sessionManager.getSessionId(); - transition.resolveNew(currentDestination); - - if (transition.detachedManager && (oldDestination.kind !== "detached" || transition.oldTarget.refs === 0)) { - void transition.detachedManager.close().catch(error => { - logger.warn("Failed to close detached bash session writer", { error: String(error) }); - }); - } - } - /** * Execute a bash command and retain the session/branch that owned its start. * @param command The bash command to execute @@ -16610,106 +6347,32 @@ export class AgentSession { * @param options.excludeFromContext If true, command output won't be sent to LLM (!! prefix) * @param options.useUserShell If true, allow caller to request configured user-shell routing */ - async executeBash( + executeBash( command: string, onChunk?: (chunk: string) => void, options?: { excludeFromContext?: boolean; useUserShell?: boolean }, ): Promise { - const target = this.#captureBashSessionTarget(); - let targetTransferred = false; - const excludeFromContext = options?.excludeFromContext === true; - const cwd = this.sessionManager.getCwd(); - - try { - if (this.#extensionRunner?.hasHandlers("user_bash")) { - const hookResult = await this.#extensionRunner.emitUserBash({ - type: "user_bash", - command, - excludeFromContext, - cwd, - }); - if (hookResult?.result) { - targetTransferred = true; - await this.#recordBashResultForTarget(target, command, hookResult.result, options); - return hookResult.result; - } - } - - const abortController = new AbortController(); - this.#bashAbortControllers.add(abortController); - let result: BashResult; - try { - result = await executeBashCommand(command, { - onChunk, - signal: abortController.signal, - sessionKey: target.sessionId, - cwd, - timeout: clampTimeout("bash", undefined, this.settings.get("tools.maxTimeout")) * 1000, - onMinimizedSave: originalText => this.#saveBashOriginalArtifact(target, originalText), - useUserShell: options?.useUserShell, - }); - } finally { - this.#bashAbortControllers.delete(abortController); - } - - targetTransferred = true; - await this.#recordBashResultForTarget(target, command, result, options); - return result; - } finally { - if (!targetTransferred) await this.#releaseBashSessionTarget(target); - } + return this.#bash.executeBash(command, onChunk, options); } /** Record a bash result supplied outside executeBash in the current ownership scope. */ recordBashResult(command: string, result: BashResult, options?: { excludeFromContext?: boolean }): void { - const target = this.#captureBashSessionTarget(); - const message = this.#createBashMessage(command, result, options); - if (this.isStreaming && target === this.#bashSessionTarget) { - this.#pendingBashMessages.push({ target, message }); - return; - } - - if (target.destination) { - try { - this.#appendBashMessage(target.destination, message); - } finally { - void this.#releaseBashSessionTarget(target); - } - return; - } - - void this.#appendOwnedBashMessage(target, message).catch(error => { - logger.error("Failed to record bash result in its owning session", { error: String(error) }); - }); + this.#bash.recordBashResult(command, result, options); } - /** - * Cancel running bash command. - */ + /** Cancel running bash commands. */ abortBash(): void { - for (const abortController of this.#bashAbortControllers) { - abortController.abort(); - } + this.#bash.abort(); } /** Whether a bash command is currently running */ get isBashRunning(): boolean { - return this.#bashAbortControllers.size > 0; + return this.#bash.isRunning; } /** Whether there are pending bash messages waiting to be flushed */ get hasPendingBashMessages(): boolean { - return this.#pendingBashMessages.length > 0; - } - - /** Flush pending bash messages after the active turn without changing their ownership. */ - async #flushPendingBashMessages(): Promise { - if (this.#pendingBashMessages.length === 0) return; - const pending = this.#pendingBashMessages; - this.#pendingBashMessages = []; - for (const { target, message } of pending) { - await this.#appendOwnedBashMessage(target, message); - } + return this.#bash.hasPendingMessages; } // ========================================================================= @@ -16723,367 +6386,70 @@ export class AgentSession { * @param onChunk Optional streaming callback for output * @param options.excludeFromContext If true, execution won't be sent to LLM ($$ prefix) */ - async executePython( + executePython( code: string, onChunk?: (chunk: string) => void, options?: { excludeFromContext?: boolean }, ): Promise { - const excludeFromContext = options?.excludeFromContext === true; - const cwd = this.sessionManager.getCwd(); - this.assertEvalExecutionAllowed(); - - const abortController = new AbortController(); - const execution = (async (): Promise => { - if (this.#extensionRunner?.hasHandlers("user_python")) { - const hookResult = await this.#extensionRunner.emitUserPython({ - type: "user_python", - code, - excludeFromContext, - cwd, - }); - this.assertEvalExecutionAllowed(); - if (hookResult?.result) { - this.recordPythonResult(code, hookResult.result, options); - return hookResult.result; - } - } - - // Use the same session ID as eval's Python backend for kernel sharing. - const sessionId = - this.getEvalSessionId() ?? - defaultEvalSessionId({ - cwd, - getSessionFile: () => this.sessionManager.getSessionFile() ?? null, - }); - const result = await executePythonCommand(code, { - cwd, - sessionId: namespacePythonSessionId(sessionId), - kernelOwnerId: this.#evalKernelOwnerId, - kernelMode: this.settings.get("python.kernelMode"), - interpreter: this.settings.get("python.interpreter")?.trim() || undefined, - onChunk, - signal: abortController.signal, - }); - this.recordPythonResult(code, result, options); - return result; - })(); - return await this.trackEvalExecution(execution, abortController); + return this.#eval.executePython(code, onChunk, options); } assertEvalExecutionAllowed(): void { - if (this.#evalExecutionDisposing) { - throw new Error("Python execution is unavailable while session disposal is in progress"); - } + this.#eval.assertExecutionAllowed(); } /** * Track Python work started outside AgentSession.executePython so dispose can await and abort it too. */ trackEvalExecution(execution: Promise, abortController: AbortController): Promise { - this.#evalAbortControllers.add(abortController); - this.#activeEvalExecutions.add(execution); - void execution.then( - () => { - this.#evalAbortControllers.delete(abortController); - this.#activeEvalExecutions.delete(execution); - }, - () => { - this.#evalAbortControllers.delete(abortController); - this.#activeEvalExecutions.delete(execution); - }, - ); - return execution; + return this.#eval.trackExecution(execution, abortController); } /** * Record a Python execution result in session history. */ recordPythonResult(code: string, result: PythonResult, options?: { excludeFromContext?: boolean }): void { - const meta = outputMeta().truncationFromSummary(result, { direction: "tail" }).get(); - const pythonMessage: PythonExecutionMessage = { - role: "pythonExecution", - code, - output: result.output, - exitCode: result.exitCode, - cancelled: result.cancelled, - truncated: result.truncated, - meta, - timestamp: Date.now(), - excludeFromContext: options?.excludeFromContext, - }; - - // If agent is streaming, defer adding to avoid breaking tool_use/tool_result ordering - if (this.isStreaming) { - this.#pendingPythonMessages.push(pythonMessage); - } else { - this.agent.appendMessage(pythonMessage); - this.sessionManager.appendMessage(pythonMessage); - } + this.#eval.recordPythonResult(code, result, options); } /** * Cancel running Python execution. */ abortEval(): void { - for (const abortController of this.#evalAbortControllers) { - abortController.abort(); - } - } - - async #waitForEvalExecutionsToSettle(timeoutMs: number): Promise { - const deadline = Date.now() + timeoutMs; - while (this.#activeEvalExecutions.size > 0) { - const remainingMs = deadline - Date.now(); - if (remainingMs <= 0) { - return false; - } - const settled = await Promise.race([ - Promise.allSettled(Array.from(this.#activeEvalExecutions)).then(() => true), - Bun.sleep(remainingMs).then(() => false), - ]); - if (!settled && this.#activeEvalExecutions.size > 0) { - return false; - } - } - return true; - } - - async #prepareEvalExecutionsForDispose(): Promise { - if (!(await this.#waitForEvalExecutionsToSettle(3_000))) { - logger.warn("Aborting active Python execution during dispose before retained kernel cleanup"); - this.abortEval(); - if (!(await this.#waitForEvalExecutionsToSettle(1_000))) { - logger.warn( - "Python execution is still active after dispose aborted all active runs; retained kernel ownership will still be detached", - ); - return false; - } - } - return true; + this.#eval.abort(); } /** Whether a Python execution is currently running */ get isEvalRunning(): boolean { - return this.#evalAbortControllers.size > 0; + return this.#eval.isRunning; } /** Whether there are pending Python messages waiting to be flushed */ get hasPendingPythonMessages(): boolean { - return this.#pendingPythonMessages.length > 0; + return this.#eval.hasPendingMessages; } /** * Flush pending Python messages to agent state and session. */ - #flushPendingPythonMessages(): void { - if (this.#pendingPythonMessages.length === 0) return; - - for (const pythonMessage of this.#pendingPythonMessages) { - this.agent.appendMessage(pythonMessage); - this.sessionManager.appendMessage(pythonMessage); - } - - this.#pendingPythonMessages = []; - } // ========================================================================= // IRC Delivery // ========================================================================= - /** - * Surfaces and consumes pending IRC incoming records before the next model - * step can inject them automatically. - * - * Tool results already expose the formatted body to the model. Leaving the - * same record in either pending IRC queue would deliver it a second time at - * the next step boundary — including on `peek`, which is why inbox peeks - * also drain here. - */ + /** Surfaces and consumes pending IRC records before automatic injection. */ drainPendingIrcInboxMessages(agentId: string, opts?: { from?: string; limit?: number }): IrcMessage[] { - const messages: IrcMessage[] = []; - const remainingInterrupts: CustomMessage[] = []; - const remainingAsides: CustomMessage[] = []; - const queues = [ - { records: this.#pendingIrcInterrupts, remaining: remainingInterrupts }, - { records: this.#pendingIrcAsides, remaining: remainingAsides }, - ]; - for (const queue of queues) { - for (const record of queue.records) { - if (record.customType !== "irc:incoming") { - queue.remaining.push(record); - continue; - } - const details = record.details; - if (!details || typeof details !== "object") { - queue.remaining.push(record); - continue; - } - const id = Reflect.get(details, "id"); - const from = Reflect.get(details, "from"); - const body = Reflect.get(details, "message"); - const replyTo = Reflect.get(details, "replyTo"); - if (typeof id !== "string" || typeof from !== "string" || typeof body !== "string") { - queue.remaining.push(record); - continue; - } - if (opts?.from !== undefined && from !== opts.from) { - queue.remaining.push(record); - continue; - } - if (opts?.limit !== undefined && messages.length >= opts.limit) { - queue.remaining.push(record); - continue; - } - messages.push({ - id, - from, - to: agentId, - body, - ts: record.timestamp, - ...(typeof replyTo === "string" ? { replyTo } : {}), - }); - } - } - this.#pendingIrcInterrupts = remainingInterrupts; - this.#pendingIrcAsides = remainingAsides; - return messages; + return this.#irc.drainInboxMessages(agentId, opts); } - /** - * Deliver an IRC message into this session (recipient side; called by the - * IrcBus). Emits the `irc_message` session event for UI cards and injects - * the rendered message into the model's context as an `irc:incoming` - * custom message: - * - * - mid-turn → queued on the aside channel and folded in at the next step - * boundary (non-interrupting, like async-result deliveries) → "injected"; - * - idle in plan mode → appended into context without waking an autonomous - * turn (convergence stays user-driven) → "injected"; - * - idle → starts a real turn with the message so the recipient wakes - * → "woken". - * - * Never blocks on the recipient's turn: the wake turn is fire-and-forget. - * - * When the sender expects a reply (`send await:true`) and this session - * cannot produce a real reply turn in time — mid-turn with async execution - * disabled (the next step boundary may be gated on the sender's own batch - * finishing), or idle in plan mode (wake turns are suppressed) — an - * ephemeral side-channel auto-reply is generated from the current context - * (the old `respondAsBackground` path) and sent back over the bus on this - * agent's behalf. - */ - async deliverIrcMessage(msg: IrcMessage, opts?: { expectsReply?: boolean }): Promise<"injected" | "woken"> { - if (this.#isDisposed) { - throw new Error("Recipient session is disposed."); - } - // Auto-reply eligibility: the sender is blocked on an answer and this - // session cannot produce a real reply turn in time — either mid-turn with - // async execution disabled (no step boundary until the sender's own batch - // ends), or idle in plan mode (autonomous wake turns are suppressed). - const planModeIdle = !this.isStreaming && this.#planModeState?.enabled === true; - const autoReply = - (opts?.expectsReply ?? false) && ((this.isStreaming && !this.settings.get("async.enabled")) || planModeIdle); - const record: CustomMessage = { - role: "custom", - customType: "irc:incoming", - content: prompt.render(ircIncomingTemplate, { - from: msg.from, - message: msg.body, - replyTo: msg.replyTo ?? "", - autoReplied: autoReply, - interrupting: this.isStreaming, - }), - display: true, - details: { id: msg.id, from: msg.from, message: msg.body, ...(msg.replyTo ? { replyTo: msg.replyTo } : {}) }, - attribution: "agent", - timestamp: msg.ts, - }; - void this.#emitSessionEvent({ type: "irc_message", message: record }); - if (this.isStreaming) { - const recipientParentId = AgentRegistry.global().get(msg.to)?.parentId; - if (recipientParentId === msg.from) { - this.agent.steer({ - role: "user", - content: prompt.render(parentIrcSteerTemplate, { from: msg.from, message: msg.body }), - attribution: "agent", - timestamp: msg.ts, - steering: true, - }); - } else { - this.#pendingIrcInterrupts.push(record); - } - if (autoReply) void this.#runIrcAutoReply(msg); - return "injected"; - } - // Plan mode: record into context but do not wake an autonomous turn. - if (this.#planModeState?.enabled) { - this.agent.appendMessage(record); - this.sessionManager.appendCustomMessageEntry( - record.customType, - record.content, - record.display, - record.details, - record.attribution ?? "agent", - ); - if (autoReply) void this.#runIrcAutoReply(msg); - return "injected"; - } - // Idle: wake a real turn so the recipient responds (shared with the stranded-aside resume). - this.#wakeForIrc([record]); - return "woken"; + /** Delivers an IRC message into this recipient session. */ + deliverIrcMessage(msg: IrcMessage, opts?: { expectsReply?: boolean }): Promise<"injected" | "woken"> { + return this.#irc.deliver(msg, opts); } - /** - * Generate and deliver an ephemeral auto-reply to `msg` on this agent's - * behalf: a no-tools side-channel turn over the current history (same - * pipeline as `/btw`), recorded into this session as an `irc:autoreply` - * aside so the model knows what was said for it, and sent back to the - * sender as a regular bus message (`replyTo: msg.id`) so their parked - * `wait`/`await:true` resolves. Failures only log — the sender then hits - * its normal wait timeout. - */ - async #runIrcAutoReply(msg: IrcMessage): Promise { - try { - const { replyText } = await this.runEphemeralTurn({ - promptText: prompt.render(ircAutoReplyTemplate, { - from: msg.from, - message: msg.body, - replyTo: msg.replyTo ?? "", - }), - }); - const body = replyText.trim(); - if (!body || this.#isDisposed) return; - const record: CustomMessage = { - role: "custom", - customType: "irc:autoreply", - content: `[IRC you → \`${msg.from}\` (auto)]\n\n${body}`, - display: true, - details: { to: msg.from, body, replyTo: msg.id }, - attribution: "agent", - timestamp: Date.now(), - }; - void this.#emitSessionEvent({ type: "irc_message", message: record }); - // Asides drain at the next step boundary; anything left over is - // flushed at the start of the next prompt (#flushPendingIrcAsides). - this.#pendingIrcAsides.push(record); - // `from` must be the id the sender addressed (msg.to) so their - // from-filtered waiter matches. - const receipt = await IrcBus.global().send({ from: msg.to, to: msg.from, body, replyTo: msg.id }); - if (receipt.outcome === "failed") { - logger.warn("IRC auto-reply delivery failed", { to: msg.from, error: receipt.error }); - } - } catch (error) { - logger.warn("IRC auto-reply turn failed", { from: msg.from, error: String(error) }); - } - } - - /** - * Emit an IRC relay observation event on this session for UI rendering only. - * Does not persist the record to history. Called by the IrcBus to surface - * agent↔agent traffic on the main session. - */ + /** Emits an IRC relay observation for UI rendering without persisting it. */ emitIrcRelayObservation(record: CustomMessage): void { - void this.#emitSessionEvent({ type: "irc_message", message: record }); + this.#irc.emitRelayObservation(record); } /** @@ -17129,7 +6495,7 @@ export class AgentSession { reasoning: toReasoningEffort(this.thinkingLevel), disableReasoning: shouldDisableReasoning(this.thinkingLevel), hideThinkingSummary: this.agent.hideThinkingSummary, - serviceTier: this.#effectiveServiceTier(model), + serviceTier: this.#models.effectiveServiceTier(model), signal: args.signal, }, model.provider, @@ -17240,25 +6606,6 @@ export class AgentSession { return messages; } - /** - * Persist any IRC asides that missed their step-boundary injection (the - * message landed after the turn's last aside drain). Called at the start - * of the next prompt so the model still sees them. - */ - #flushPendingIrcAsides(): void { - if (this.#pendingIrcInterrupts.length === 0 && this.#pendingIrcAsides.length === 0) return; - const records = [...this.#pendingIrcInterrupts, ...this.#pendingIrcAsides]; - this.#pendingIrcInterrupts = []; - this.#pendingIrcAsides = []; - for (const record of records) { - // emitExternalEvent on message_end appends to agent state and dispatches - // to all session listeners, which in turn handle TUI rendering and - // sessionManager persistence via #handleAgentEvent. - this.agent.emitExternalEvent({ type: "message_start", message: record }); - this.agent.emitExternalEvent({ type: "message_end", message: record }); - } - } - // ========================================================================= // Session Management // ========================================================================= @@ -17303,11 +6650,11 @@ export class AgentSession { await this.abort({ goalReason: "internal" }); await this.#sessionBeforeSwitchReconciler?.(); - await this.#flushPendingBashMessages(); + await this.#bash.flushPending(); // Flush pending writes before switching so restore snapshots reflect committed state. await this.sessionManager.flush(); const previousSessionState = this.sessionManager.captureState(); - const bashTransition = this.#beginBashSessionTransition(); + const bashTransition = this.#bash.beginSessionTransition(); // Only same-session reloads compare against the prior context to detect // rollback edits (`#didSessionMessagesChange` below). Building it for a // different-session switch is a pure waste — and on huge pre-fix sessions @@ -17326,14 +6673,14 @@ export class AgentSession { const previousPendingNextTurnMessages = [...this.#pendingNextTurnMessages]; const previousScheduledHiddenNextTurnGeneration = this.#scheduledHiddenNextTurnGeneration; const previousModel = this.model; - const previousThinkingLevel = this.#thinkingLevel; - const previousAutoThinking = this.#autoThinking; - const previousAutoResolvedLevel = this.#autoResolvedLevel; - const previousServiceTierByFamily = this.#serviceTierByFamily; + const previousThinkingLevel = this.thinkingLevel; + const previousAutoThinking = this.isAutoThinking; + const previousAutoResolvedLevel = this.autoResolvedThinkingLevel(); + const previousServiceTierByFamily = this.serviceTierByFamily; const previousTools = [...this.agent.state.tools]; - const previousBaseSystemPrompt = this.#baseSystemPrompt; + const previousBaseSystemPrompt = this.#tools.baseSystemPrompt; const previousSystemPrompt = this.agent.state.systemPrompt; - const previousBaseSystemPromptBeforeMemoryPromotion = this.#baseSystemPromptBeforeMemoryPromotion; + const previousBaseSystemPromptBeforeMemoryPromotion = this.#memory.promotionSnapshot; const previousFreshProviderSessionId = this.#freshProviderSessionId; const previousInheritedProviderPromptCacheKey = this.#inheritedProviderPromptCacheKey; @@ -17352,20 +6699,19 @@ export class AgentSession { try { await this.sessionManager.setSessionFile(sessionPath); - this.#markBashSessionTransition(bashTransition); + this.#bash.markSessionTransition(bashTransition); if (switchingToDifferentSession) { this.#freshProviderSessionId = undefined; this.#clearInheritedProviderPromptCacheKey(); this.#adoptInheritedProviderPromptCacheKey(); } this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); + this.#memory.rekeyForCurrentSessionId(); let sessionContext = this.buildDisplaySessionContext(); const didReloadConversationChange = previousSessionContext !== undefined && - this.#didSessionMessagesChange(previousSessionContext.messages, sessionContext.messages); + didSessionMessagesChange(previousSessionContext.messages, sessionContext.messages); this.#rehydrateCheckpointRewindState(); // Emit session_switch event to hooks @@ -17378,8 +6724,8 @@ export class AgentSession { } this.agent.replaceMessages(sessionContext.messages); - this.#resetAdvisorSessionState(); - this.#syncTodoPhasesFromBranch(); + this.#advisors.resetSessionState(); + this.#todo.syncFromBranch(); if (switchingToDifferentSession) { this.#closeAllProviderSessions("session switch"); } else if (didReloadConversationChange) { @@ -17456,26 +6802,13 @@ export class AgentSession { ? AUTO_THINKING : (sessionContext.thinkingLevel as ThinkingLevel | undefined) : defaultThinkingLevel; - if (restoredThinkingLevel === AUTO_THINKING) { - this.#autoThinking = true; - // Resume in auto (pending) like a fresh auto session: the next user - // turn reclassifies. We intentionally do not seed the last resolved - // effort, so the cold (--continue) and in-app switch paths display - // identically as `auto` until then. - this.#autoResolvedLevel = undefined; - this.#thinkingLevel = resolveProvisionalAutoLevel(this.model); - } else { - this.#autoThinking = false; - this.#autoResolvedLevel = undefined; - this.#thinkingLevel = resolveThinkingLevelForModel(this.model, restoredThinkingLevel); - } - this.#applyThinkingLevelToAgent(this.#thinkingLevel); - this.#serviceTierByFamily = hasServiceTierEntry - ? (sessionContext.serviceTier ?? {}) - : configuredServiceTierByFamily; + this.#models.restoreThinkingLevel(restoredThinkingLevel); + this.#models.restoreServiceTiers( + hasServiceTierEntry ? (sessionContext.serviceTier ?? {}) : configuredServiceTierByFamily, + ); if (switchingToDifferentSession) { - await this.#resetMemoryContextForNewTranscript(); + await this.#memory.resetContextForNewTranscript(); } this.#reconnectToAgent(); try { @@ -17497,17 +6830,16 @@ export class AgentSession { error: String(refreshErr), }); } - this.#finishBashSessionTransition(bashTransition, true); + this.#bash.finishSessionTransition(bashTransition, true); return true; } catch (error) { this.sessionManager.restoreState(previousSessionState); this.#freshProviderSessionId = previousFreshProviderSessionId; this.#syncAgentSessionId(previousSessionState.sessionId); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); + this.#memory.rekeyForCurrentSessionId(); this.agent.setTools(previousTools); - this.#baseSystemPrompt = previousBaseSystemPrompt; - this.#baseSystemPromptBeforeMemoryPromotion = previousBaseSystemPromptBeforeMemoryPromotion; + this.#tools.setBaseSystemPrompt(previousBaseSystemPrompt); + this.#memory.restorePromotionSnapshot(previousBaseSystemPromptBeforeMemoryPromotion); this.agent.setSystemPrompt(previousSystemPrompt); this.agent.replaceMessages(previousAgentMessages); this.agent.replaceQueues(previousSteeringMessages, previousFollowUpMessages); @@ -17521,13 +6853,10 @@ export class AgentSession { if (previousModel) { this.agent.setModel(previousModel); } - this.#thinkingLevel = previousThinkingLevel; - this.#autoThinking = previousAutoThinking; - this.#autoResolvedLevel = previousAutoResolvedLevel; - this.#applyThinkingLevelToAgent(previousThinkingLevel); - this.#serviceTierByFamily = previousServiceTierByFamily; - this.#syncTodoPhasesFromBranch(); - this.#resetAllAdvisorRuntimes(); + this.#models.restoreThinkingSnapshot(previousThinkingLevel, previousAutoThinking, previousAutoResolvedLevel); + this.#models.restoreServiceTiers(previousServiceTierByFamily); + this.#todo.syncFromBranch(); + this.#advisors.resetAllRuntimes(); this.#reconnectToAgent(); try { await this.#sessionSwitchReconciler?.(); @@ -17537,7 +6866,7 @@ export class AgentSession { error: String(reconcileError), }); } - this.#finishBashSessionTransition(bashTransition, false); + this.#bash.finishSessionTransition(bashTransition, false); throw error; } } @@ -17583,10 +6912,10 @@ export class AgentSession { this.#pendingNextTurnMessages = []; this.#scheduledHiddenNextTurnGeneration = undefined; - await this.#flushPendingBashMessages(); + await this.#bash.flushPending(); // Flush pending writes before branching await this.sessionManager.flush(); - const bashTransition = this.#beginBashSessionTransition(); + const bashTransition = this.#bash.beginSessionTransition(); this.#cancelOwnAsyncJobs(); this.#abortAutolearnCapture(); await this.#drainAutolearnCapture(); @@ -17598,19 +6927,18 @@ export class AgentSession { } else { this.sessionManager.createBranchedSession(selectedEntry.parentId); } - this.#markBashSessionTransition(bashTransition); + this.#bash.markSessionTransition(bashTransition); sessionTransitioned = true; } finally { - this.#finishBashSessionTransition(bashTransition, sessionTransitioned); + this.#bash.finishSessionTransition(bashTransition, sessionTransitioned); } this.#rehydrateCheckpointRewindState(); - this.#syncTodoPhasesFromBranch(); + this.#todo.syncFromBranch(); this.#freshProviderSessionId = undefined; this.#clearInheritedProviderPromptCacheKey(); this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); - await this.#resetMemoryContextForNewTranscript(); + this.#memory.rekeyForCurrentSessionId(); + await this.#memory.resetContextForNewTranscript(); // Reload messages from entries (works for both file and in-memory mode) const sessionContext = this.buildDisplaySessionContext(); @@ -17625,7 +6953,7 @@ export class AgentSession { if (!skipConversationRestore) { this.agent.replaceMessages(sessionContext.messages); - this.#resetAdvisorSessionState(); + this.#advisors.resetSessionState(); this.#closeCodexProviderSessionsForHistoryRewrite(); } @@ -17685,9 +7013,9 @@ export class AgentSession { await this.abort({ goalReason: "internal", reason: "branching /btw" }); this.agent.replaceQueues([], []); } - await this.#flushPendingBashMessages(); + await this.#bash.flushPending(); await this.sessionManager.flush(); - const bashTransition = this.#beginBashSessionTransition(); + const bashTransition = this.#bash.beginSessionTransition(); this.#cancelOwnAsyncJobs(); this.#abortAutolearnCapture(); await this.#drainAutolearnCapture(); @@ -17695,10 +7023,10 @@ export class AgentSession { let sessionTransitioned = false; try { this.sessionManager.createBranchedSession(leafId); - this.#markBashSessionTransition(bashTransition); + this.#bash.markSessionTransition(bashTransition); sessionTransitioned = true; } finally { - this.#finishBashSessionTransition(bashTransition, sessionTransitioned); + this.#bash.finishSessionTransition(bashTransition, sessionTransitioned); } this.#rehydrateCheckpointRewindState(); @@ -17708,12 +7036,11 @@ export class AgentSession { timestamp: Date.now(), }); this.sessionManager.appendMessage(sanitizeAssistantForReparentedHistory(assistantMessage)); - this.#syncTodoPhasesFromBranch(); + this.#todo.syncFromBranch(); this.#freshProviderSessionId = undefined; this.#syncAgentSessionId(); - this.#rekeyHindsightMemoryForCurrentSessionId(); - this.#rekeyMnemopiMemoryForCurrentSessionId(); - await this.#resetMemoryContextForNewTranscript(); + this.#memory.rekeyForCurrentSessionId(); + await this.#memory.resetContextForNewTranscript(); const sessionContext = this.buildDisplaySessionContext(); @@ -17725,7 +7052,7 @@ export class AgentSession { } this.agent.replaceMessages(sessionContext.messages); - this.#resetAdvisorSessionState(); + this.#advisors.resetSessionState(); this.#closeCodexProviderSessionsForHistoryRewrite(); return { cancelled: false, sessionFile: this.sessionFile }; @@ -17787,7 +7114,7 @@ export class AgentSession { */ reopenAsk?: { toolCallId: string; questions: AskToolInput["questions"] }; }> { - await this.#flushPendingBashMessages(); + await this.#bash.flushPending(); const oldLeafId = this.sessionManager.getLeafId(); const targetEntry = this.sessionManager.getEntry(targetId); @@ -17982,7 +7309,7 @@ export class AgentSession { // Switch leaf (with or without summary) // Summary is attached at the navigation target position (newLeafId), not the old branch - const bashTransition = this.#beginBashSessionTransition(); + const bashTransition = this.#bash.beginSessionTransition(); let summaryEntry: BranchSummaryEntry | undefined; let branchTransitioned = false; try { @@ -18000,10 +7327,10 @@ export class AgentSession { } else { this.sessionManager.branch(newLeafId); } - this.#markBashSessionTransition(bashTransition); + this.#bash.markSessionTransition(bashTransition); branchTransitioned = true; } finally { - this.#finishBashSessionTransition(bashTransition, branchTransitioned); + this.#bash.finishSessionTransition(bashTransition, branchTransitioned); } // Update agent state — build display context to populate agent messages. @@ -18011,8 +7338,8 @@ export class AgentSession { const displayContext = deobfuscateSessionContext(stateContext, this.#obfuscator); this.agent.replaceMessages(displayContext.messages); this.#rehydrateCheckpointRewindState(); - this.#resetAdvisorSessionState(); - this.#syncTodoPhasesFromBranch(); + this.#advisors.resetSessionState(); + this.#todo.syncFromBranch(); this.#closeCodexProviderSessionsForHistoryRewrite(); this.#branchSummaryAbortController = undefined; @@ -18136,78 +7463,7 @@ export class AgentSession { * Get session statistics. */ getSessionStats(): SessionStats { - const state = this.state; - const userMessages = state.messages.filter(m => m.role === "user").length; - const assistantMessages = state.messages.filter(m => m.role === "assistant").length; - const toolResults = state.messages.filter(m => m.role === "toolResult").length; - - let toolCalls = 0; - let totalInput = 0; - let totalOutput = 0; - let totalCacheRead = 0; - let totalReasoning = 0; - let totalCacheWrite = 0; - let totalTokens = 0; - let totalCost = 0; - let totalPremiumRequests = 0; - - const getTaskToolUsage = (details: unknown): Usage | undefined => { - if (!details || typeof details !== "object") return undefined; - const record = details as Record; - const usage = record.usage; - if (!usage || typeof usage !== "object") return undefined; - return usage as Usage; - }; - - for (const message of state.messages) { - if (message.role === "assistant") { - const assistantMsg = message as AssistantMessage; - toolCalls += assistantMsg.content.filter(c => c.type === "toolCall").length; - totalInput += assistantMsg.usage.input; - totalOutput += assistantMsg.usage.output; - totalReasoning += assistantMsg.usage.reasoningTokens ?? 0; - totalCacheRead += assistantMsg.usage.cacheRead; - totalCacheWrite += assistantMsg.usage.cacheWrite; - totalTokens += assistantMsg.usage.totalTokens; - totalPremiumRequests += assistantMsg.usage.premiumRequests ?? 0; - totalCost += assistantMsg.usage.cost.total; - } - - if (message.role === "toolResult" && message.toolName === "task") { - const usage = getTaskToolUsage(message.details); - if (usage) { - totalInput += usage.input; - totalOutput += usage.output; - totalReasoning += usage.reasoningTokens ?? 0; - totalCacheRead += usage.cacheRead; - totalCacheWrite += usage.cacheWrite; - totalTokens += usage.totalTokens; - totalPremiumRequests += usage.premiumRequests ?? 0; - totalCost += usage.cost.total; - } - } - } - - return { - sessionFile: this.sessionFile, - sessionId: this.sessionId, - userMessages, - assistantMessages, - toolCalls, - toolResults, - totalMessages: state.messages.length, - tokens: { - input: totalInput, - output: totalOutput, - reasoning: totalReasoning, - cacheRead: totalCacheRead, - cacheWrite: totalCacheWrite, - total: totalTokens, - }, - cost: totalCost, - premiumRequests: totalPremiumRequests, - contextUsage: this.getContextUsage(), - }; + return this.#stats.getSessionStats(); } /** @@ -18219,151 +7475,11 @@ export class AgentSession { contextWindow?: number; pendingMessages?: AgentMessage[]; }): ContextUsageBreakdown | undefined { - const model = this.model; - const rawContextWindow = options?.contextWindow ?? model?.contextWindow ?? 0; - const contextWindow = Number.isFinite(rawContextWindow) && rawContextWindow > 0 ? rawContextWindow : 0; - - const { skillsTokens, toolsTokens, systemContextTokens, systemPromptTokens } = computeNonMessageBreakdown(this); - const categoryNonMessageTokens = skillsTokens + toolsTokens + systemContextTokens + systemPromptTokens; - const currentNonMessageTokens = computeNonMessageTokens(this); - - const branchEntries = this.sessionManager.getBranch(); - const latestCompaction = getLatestCompactionEntry(branchEntries); - const compactionIndex = latestCompaction ? branchEntries.lastIndexOf(latestCompaction) : -1; - - let usedTokens = 0; - let anchored = false; - - const pendingMessages = options?.pendingMessages ?? []; - - const pending = this.#pendingContextSnapshot; - - // Always locate the latest real assistant-usage anchor after the last - // compaction. Its provider-reported promptTokens is ground truth for - // everything up to that point; only the tail after it is estimated. - let anchorEntry: SessionMessageEntry | undefined; - for (let i = branchEntries.length - 1; i > compactionIndex; i--) { - const entry = branchEntries[i]; - if (entry.type === "message" && entry.message.role === "assistant") { - const assistant = entry.message; - if (assistant.stopReason !== "aborted" && assistant.stopReason !== "error" && assistant.usage) { - anchorEntry = entry; - break; - } - } - } - - const resolvedActiveMessages = this.messages; - let resolvedAnchorIndex = -1; - let anchorAssistant: AssistantMessage | undefined; - if (anchorEntry) { - const a = anchorEntry.message as AssistantMessage; - anchorAssistant = a; - resolvedAnchorIndex = resolvedActiveMessages.indexOf(a); - if (resolvedAnchorIndex === -1) { - resolvedAnchorIndex = resolvedActiveMessages.findIndex( - msg => msg.role === "assistant" && msg.timestamp === a.timestamp, - ); - } - } - - // A real anchor supersedes the in-flight estimate only once a step of the - // CURRENT turn has produced provider usage — i.e. it resolves at or after - // the pending cutoff. While the turn's first response is still pending (or - // the newest real anchor predates this turn) the pending snapshot is the - // only thing accounting for the just-submitted prompt, so it wins. This - // keeps a long tool turn from stacking an estimate of the entire tail on - // top of a stale turn-start prompt. - const useAnchor = - anchorAssistant !== undefined && - resolvedAnchorIndex !== -1 && - (!pending || resolvedAnchorIndex >= pending.cutoffCount); - - if (useAnchor && anchorAssistant) { - const promptTokens = - anchorAssistant.contextSnapshot?.promptTokens ?? calculatePromptTokens(anchorAssistant.usage); - const nonMessageTokens = anchorAssistant.contextSnapshot?.nonMessageTokens ?? computeNonMessageTokens(this); - anchored = true; - let tailTokens = 0; - for (let i = resolvedAnchorIndex + 1; i < resolvedActiveMessages.length; i++) { - tailTokens += estimateTokens(resolvedActiveMessages[i]); - } - usedTokens = - promptTokens + - Math.max(0, currentNonMessageTokens - nonMessageTokens) + - tailTokens + - pendingMessages.reduce((sum, msg) => sum + estimateTokens(msg), 0); - } else if (pending) { - anchored = true; - let tailTokens = 0; - if (resolvedActiveMessages.length > pending.cutoffCount) { - for (let i = pending.cutoffCount; i < resolvedActiveMessages.length; i++) { - tailTokens += estimateTokens(resolvedActiveMessages[i]); - } - } - usedTokens = - pending.promptTokens + - Math.max(0, currentNonMessageTokens - pending.nonMessageTokens) + - tailTokens + - pendingMessages.reduce((sum, msg) => sum + estimateTokens(msg), 0); - } - - if (!anchored && !pending && branchEntries.length === 0) { - // Fallback: look for the latest assistant message with usage/snapshot in this.messages (for branchless/fake sessions in tests) - for (let i = resolvedActiveMessages.length - 1; i >= 0; i--) { - const msg = resolvedActiveMessages[i]; - if (msg.role === "assistant" && msg.stopReason !== "aborted" && msg.stopReason !== "error" && msg.usage) { - const promptTokens = msg.contextSnapshot?.promptTokens ?? calculatePromptTokens(msg.usage); - const nonMessageTokens = msg.contextSnapshot?.nonMessageTokens ?? computeNonMessageTokens(this); - - let tailTokens = 0; - for (let j = i + 1; j < resolvedActiveMessages.length; j++) { - tailTokens += estimateTokens(resolvedActiveMessages[j]); - } - - usedTokens = - promptTokens + - Math.max(0, currentNonMessageTokens - nonMessageTokens) + - tailTokens + - pendingMessages.reduce((sum, msg) => sum + estimateTokens(msg), 0); - anchored = true; - break; - } - } - } - if (!anchored) { - let messagesTokens = 0; - for (const msg of resolvedActiveMessages) { - messagesTokens += estimateTokens(msg); - } - usedTokens = - currentNonMessageTokens + - messagesTokens + - pendingMessages.reduce((sum, msg) => sum + estimateTokens(msg), 0); - } - - const messagesTokens = Math.max(0, usedTokens - categoryNonMessageTokens); - - return { - contextWindow, - anchored, - usedTokens, - systemPromptTokens, - systemToolsTokens: toolsTokens, - systemContextTokens, - skillsTokens, - messagesTokens, - }; + return this.#stats.getContextBreakdown(options); } getContextUsage(options?: { contextWindow?: number }): ContextUsage | undefined { - const breakdown = this.getContextBreakdown(options); - if (!breakdown) return undefined; - return { - tokens: breakdown.usedTokens, - contextWindow: breakdown.contextWindow, - percent: breakdown.contextWindow > 0 ? (breakdown.usedTokens / breakdown.contextWindow) * 100 : 0, - }; + return this.#stats.getContextUsage(options); } /** @@ -18372,46 +7488,7 @@ export class AgentSession { * a value computed mid-turn cannot persist after the turn ends/aborts. */ get contextUsageRevision(): number { - return this.#contextUsageRevision; - } - - #setPendingContextSnapshot( - snapshot: { promptTokens: number; nonMessageTokens: number; cutoffCount: number } | undefined, - ): void { - this.#pendingContextSnapshot = snapshot; - this.#contextUsageRevision++; - } - - /** - * Rebase the in-flight pending context snapshot onto the current message - * set after a compaction (or its dead-end rescue) rewrote history mid-run. - * The snapshot captures the prompt as submitted at run start and lives for - * the whole run; once a compaction entry lands, every earlier usage anchor - * is hidden from {@link getContextBreakdown}, so the stale run-start figure - * would be reported as live context until the next provider response. That - * inflated residual is what the post-compaction headroom/retry-fit checks - * measure — a run that started above the recovery band then trips the - * "freed too little context" dead-end even when compaction genuinely - * shrank the context. No-op while no prompt is in flight. - */ - #rebasePendingContextSnapshotAfterCompaction(): void { - if (!this.#pendingContextSnapshot) return; - const nonMessageTokens = computeNonMessageTokens(this); - this.#setPendingContextSnapshot({ - promptTokens: nonMessageTokens + this.messages.reduce((sum, msg) => sum + estimateTokens(msg), 0), - nonMessageTokens, - cutoffCount: this.messages.length, - }); - } - - #ingestProviderUsageHeaders(response: ProviderResponseMetadata, model?: Model): void { - const provider = model?.provider; - if (!provider) return; - // No-op for providers whose usage strategy lacks a header parser. - this.#modelRegistry.authStorage.ingestUsageHeaders(provider, response.headers, { - sessionId: this.agent.sessionId, - baseUrl: this.#modelRegistry.getProviderBaseUrl?.(provider), - }); + return this.#stats.revision; } async fetchUsageReports(signal?: AbortSignal): Promise { @@ -18699,7 +7776,7 @@ export class AgentSession { messages: this.messages, systemPrompt: this.agent.state.systemPrompt, model: this.agent.state.model, - thinkingLevel: this.#thinkingLevel, + thinkingLevel: this.thinkingLevel, tools: this.agent.state.tools, inlineToolDescriptors: this.#pruneToolDescriptions, }); @@ -18723,8 +7800,8 @@ export class AgentSession { const llmMessages = await this.convertMessagesToLlm(messages); const payload = { model: this.agent.state.model ?? null, - thinkingLevel: this.#thinkingLevel ?? null, - serviceTier: this.#serviceTierEntry(), + thinkingLevel: this.thinkingLevel ?? null, + serviceTier: this.#models.serviceTierEntry(), systemPrompt: this.agent.state.systemPrompt, tools: this.agent.state.tools.map(tool => ({ name: tool.name, @@ -18747,13 +7824,7 @@ export class AgentSession { * @returns true when the advisor is actively running after the call. */ setAdvisorEnabled(enabled: boolean): boolean { - this.#advisorEnabled = enabled; - if (enabled) { - if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime(); - return this.#buildAdvisorRuntime(true); - } - this.#stopAdvisorRuntime(); - return false; + return this.#advisors.setAdvisorEnabled(enabled); } /** @@ -18762,7 +7833,7 @@ export class AgentSession { * @returns true when the advisor is actively running after the call. */ toggleAdvisorEnabled(): boolean { - return this.setAdvisorEnabled(!this.#advisorEnabled); + return this.#advisors.toggleAdvisorEnabled(); } /** @@ -18774,19 +7845,14 @@ export class AgentSession { * @returns the number of advisors active after the rebuild. */ applyAdvisorConfigs(advisors: AdvisorConfig[], sharedInstructions: string | undefined): number { - this.#advisorConfigs = advisors; - this.#advisorSharedInstructions = sharedInstructions; - if (!this.#advisorEnabled) return 0; - this.#stopAdvisorRuntime(); - this.#buildAdvisorRuntime(true); - return this.#advisors.length; + return this.#advisors.applyAdvisorConfigs(advisors, sharedInstructions); } /** * Whether the advisor setting is enabled for this session. */ isAdvisorEnabled(): boolean { - return this.#advisorEnabled; + return this.#advisors.isAdvisorEnabled(); } /** @@ -18796,7 +7862,7 @@ export class AgentSession { * not merely the setting. Drives the status-line badge and `/dump advisor`. */ isAdvisorActive(): boolean { - return this.#advisors.length > 0; + return this.#advisors.isAdvisorActive(); } /** @@ -18806,7 +7872,7 @@ export class AgentSession { * no servers) is absent. */ getAdvisorAvailableToolNames(): string[] { - return (this.#advisorTools ?? []).map(tool => tool.name); + return this.#advisors.getAdvisorAvailableToolNames(); } /** @@ -18817,7 +7883,7 @@ export class AgentSession { * (`streamFn`, `promptCacheKey`, `providerSessionState`, ...). */ getAdvisorAgent(): Agent | undefined { - return this.#advisors[0]?.agent; + return this.#advisors.getAdvisorAgent(); } /** @@ -18826,229 +7892,20 @@ export class AgentSession { * Avoids re-tokenizing the advisor transcript on every render frame. */ getAdvisorStatusOverview(): { configured: boolean; advisors: { name: string; status: AdvisorRuntimeStatus }[] } { - // Override stale map entries with live runtime status: failureNotified/quotaExhausted - // clear on reset() but #advisorStatuses lags until the next build. - const liveStatusBySlug = new Map(); - for (const a of this.#advisors) { - liveStatusBySlug.set( - a.slug, - a.runtime.quotaExhausted ? "quota_exhausted" : a.runtime.failureNotified ? "error" : "running", - ); - } - const advisors = [...this.#advisorStatuses.entries()].map(([slug, { name, status }]) => ({ - name, - status: liveStatusBySlug.get(slug) ?? status, - })); - return { configured: this.#advisorEnabled, advisors }; + return this.#advisors.getAdvisorStatusOverview(); } /** * Return structured advisor stats for the status command and TUI panel. */ getAdvisorStats(): AdvisorStats { - const configured = this.#advisorEnabled; - const liveAdvisors = this.#advisors.map(a => this.#computeAdvisorStat(a)); - // Build the complete roster from #advisorStatuses, which already has the - // correct de-duped slugs as keys. Live advisors (from #advisors) carry full - // token/cost data; disabled/no-model/quota-exhausted advisors appear as - // skeleton entries with just name + status so the status line renders a dot. - const liveStatBySlug = new Map(this.#advisors.map((a, i) => [a.slug, liveAdvisors[i]])); - const roster: PerAdvisorStat[] = []; - for (const [slug, entry] of this.#advisorStatuses) { - const live = liveStatBySlug.get(slug); - if (live) { - roster.push(live); - } else { - roster.push({ - name: entry.name, - status: entry.status, - contextWindow: 0, - contextTokens: 0, - tokens: { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - cost: 0, - messages: { user: 0, assistant: 0, total: 0 }, - }); - } - } - const active = liveAdvisors.length > 0; - if (liveAdvisors.length === 0) { - return { - configured, - active, - contextWindow: 0, - contextTokens: 0, - tokens: { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - cost: 0, - messages: { user: 0, assistant: 0, total: 0 }, - advisors: roster, - }; - } - const tokens = { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }; - const messages = { user: 0, assistant: 0, total: 0 }; - let cost = 0; - let contextTokens = 0; - for (const a of liveAdvisors) { - tokens.input += a.tokens.input; - tokens.output += a.tokens.output; - tokens.reasoning += a.tokens.reasoning; - tokens.cacheRead += a.tokens.cacheRead; - tokens.cacheWrite += a.tokens.cacheWrite; - tokens.total += a.tokens.total; - messages.user += a.messages.user; - messages.assistant += a.messages.assistant; - messages.total += a.messages.total; - cost += a.cost; - contextTokens += a.contextTokens; - } - // Single-advisor displays read the top-level model/window directly; surface the - // first advisor's so the legacy status line stays byte-identical. - return { - configured, - active, - model: liveAdvisors[0].model, - contextWindow: liveAdvisors[0].contextWindow, - contextTokens, - tokens, - cost, - messages, - advisors: roster, - }; - } - - /** Compute one advisor's stats slice (tokens, cost, context, message counts). */ - #computeAdvisorStat(advisor: ActiveAdvisor): PerAdvisorStat { - const model = advisor.agent.state.model; - const messages = advisor.agent.state.messages; - const contextTokens = this.#estimateAdvisorContextTokens(messages); - let input = 0; - let output = 0; - let reasoning = 0; - let cacheRead = 0; - let cacheWrite = 0; - let totalTokens = 0; - let cost = 0; - let user = 0; - let assistant = 0; - for (const message of messages) { - if (message.role === "user") user++; - if (message.role === "assistant") { - assistant++; - const assistantMsg = message as AssistantMessage; - input += assistantMsg.usage.input; - output += assistantMsg.usage.output; - reasoning += assistantMsg.usage.reasoningTokens ?? 0; - cacheRead += assistantMsg.usage.cacheRead; - cacheWrite += assistantMsg.usage.cacheWrite; - totalTokens += assistantMsg.usage.totalTokens; - cost += assistantMsg.usage.cost.total; - } - } - return { - name: advisor.name, - status: advisor.runtime.quotaExhausted - ? "quota_exhausted" - : advisor.runtime.failureNotified - ? "error" - : "running", - model, - contextWindow: model.contextWindow ?? 0, - contextTokens, - tokens: { input, output, reasoning, cacheRead, cacheWrite, total: totalTokens }, - cost, - messages: { user, assistant, total: messages.length }, - sessionId: advisor.agent.sessionId, - }; + return this.#advisors.getAdvisorStats(); } /** * Format a concise advisor status line for ACP/text output. */ formatAdvisorStatus(): string { - const stats = this.getAdvisorStats(); - if (!stats.active && stats.advisors.length === 0) { - return stats.configured - ? "Advisor setting is enabled, but no model is assigned to the 'advisor' role." - : "Advisor is disabled."; - } - if (stats.advisors.length <= 1) { - const s = stats.advisors[0]; - if (s && s.status === "no_model") { - return stats.configured - ? "Advisor setting is enabled, but no model is assigned to the 'advisor' role." - : "Advisor is disabled."; - } - const contextLine = - s.contextWindow > 0 - ? `Context: ${s.contextTokens.toLocaleString()} / ${s.contextWindow.toLocaleString()} tokens (${Math.round((s.contextTokens / s.contextWindow) * 100)}%)` - : `Context: ${s.contextTokens.toLocaleString()} tokens`; - const spendParts = [`${s.tokens.input.toLocaleString()} input`, `${s.tokens.output.toLocaleString()} output`]; - if (s.tokens.cacheRead > 0) spendParts.push(`${s.tokens.cacheRead.toLocaleString()} cache read`); - if (s.tokens.cacheWrite > 0) spendParts.push(`${s.tokens.cacheWrite.toLocaleString()} cache write`); - const spendLine = `Spend: ${spendParts.join(", ")}, $${s.cost.toFixed(4)}`; - if (!s.model || s.status !== "running") return `Advisor "${s.name}" is ${s.status.replace("_", " ")}.`; - return `Advisor is enabled (${s.model.provider}/${s.model.id}). ${contextLine}. ${spendLine}.`; - } - const lines = [`Advisors enabled (${stats.advisors.length}):`]; - for (const s of stats.advisors) { - const ctx = - s.contextWindow > 0 - ? `${s.contextTokens.toLocaleString()} / ${s.contextWindow.toLocaleString()} (${Math.round((s.contextTokens / s.contextWindow) * 100)}%)` - : `${s.contextTokens.toLocaleString()}`; - lines.push( - ` • ${s.name}${s.model && s.status === "running" ? ` (${s.model.provider}/${s.model.id})` : ` [${s.status}]`} — context ${ctx} tokens, $${s.cost.toFixed(4)}`, - ); - } - lines.push( - `Totals: ${stats.tokens.input.toLocaleString()} input, ${stats.tokens.output.toLocaleString()} output, $${stats.cost.toFixed(4)}.`, - ); - return lines.join("\n"); - } - - /** - * Estimate the advisor's current context tokens. A successful provider usage - * after the latest advisor compaction is ground truth for the prompt plus its - * generated output; only messages after that anchor are estimated. Usage from - * retained pre-compaction messages is stale and must not immediately retrigger - * maintenance on the newly compacted context. - */ - #estimateAdvisorContextTokens(messages: AgentMessage[]): number { - let usageAnchorStartIndex = 0; - for (let i = messages.length - 1; i >= 0; i--) { - const message = messages[i]; - if (message.role !== "compactionSummary") continue; - const advisorSummary = message as AdvisorCompactionSummaryMessage; - // Advisor summaries created before this runtime-only boundary existed have - // no trustworthy way to distinguish retained from newly appended messages. - // Conservatively ignore every current assistant until the next compaction. - usageAnchorStartIndex = advisorSummary.advisorUsageAnchorStartIndex ?? messages.length; - break; - } - - let lastUsageIndex: number | undefined; - let lastUsage: AssistantMessage["usage"] | undefined; - for (let i = messages.length - 1; i >= usageAnchorStartIndex; i--) { - const message = messages[i]; - if (message.role !== "assistant") continue; - const assistant = message as AssistantMessage; - if (assistant.stopReason !== "aborted" && assistant.stopReason !== "error" && assistant.usage) { - lastUsage = assistant.usage; - lastUsageIndex = i; - break; - } - } - - const estimateOptions = { excludeEncryptedReasoning: true } as const; - if (!lastUsage || lastUsageIndex === undefined) { - let estimated = 0; - for (const message of messages) { - estimated += estimateTokens(message, estimateOptions); - } - return estimated; - } - let trailingTokens = 0; - for (let i = lastUsageIndex + 1; i < messages.length; i++) { - trailingTokens += estimateTokens(messages[i], estimateOptions); - } - return calculateContextTokens(lastUsage) + trailingTokens; + return this.#advisors.formatAdvisorStatus(); } /** @@ -19058,21 +7915,7 @@ export class AgentSession { * {@link formatSessionAsText}. Returns null when no advisor is active. */ formatAdvisorHistoryAsText(options?: { compact?: boolean }): string | null { - if (this.#advisors.length === 0) return null; - const dump = (a: ActiveAdvisor): string => - options?.compact - ? formatSessionHistoryMarkdown(a.agent.state.messages) - : formatSessionDumpText({ - messages: a.agent.state.messages, - systemPrompt: a.agent.state.systemPrompt, - model: a.agent.state.model, - thinkingLevel: a.agent.state.thinkingLevel, - tools: a.agent.state.tools, - }); - if (this.#advisors.length === 1) return dump(this.#advisors[0]); - return this.#advisors - .map(a => `### Advisor: ${a.name} (${a.agent.state.model.provider}/${a.agent.state.model.id})\n\n${dump(a)}`) - .join("\n\n"); + return this.#advisors.formatAdvisorHistoryAsText(options); } // ========================================================================= diff --git a/packages/coding-agent/src/session/bash-runner.ts b/packages/coding-agent/src/session/bash-runner.ts new file mode 100644 index 000000000..7ba065d8e --- /dev/null +++ b/packages/coding-agent/src/session/bash-runner.ts @@ -0,0 +1,326 @@ +import * as path from "node:path"; +import type { Agent } from "@oh-my-pi/pi-agent-core"; +import { logger } from "@oh-my-pi/pi-utils"; +import type { Settings } from "../config/settings"; +import { type BashResult, executeBash as executeBashCommand } from "../exec/bash-executor"; +import type { ExtensionRunner } from "../extensibility/extensions"; +import { outputMeta } from "../tools/output-meta"; +import { clampTimeout } from "../tools/tool-timeouts"; +import type { BashExecutionMessage } from "./messages"; +import type { SessionManager } from "./session-manager"; + +/** Destination that owns a bash result after a session or branch transition. */ +export type BashAppendDestination = + | { kind: "current"; manager: SessionManager } + | { kind: "detached"; manager: SessionManager } + | { kind: "branch"; manager: SessionManager; parentId: string | null }; + +/** Reference-counted session target captured when a bash execution starts. */ +export interface BashSessionTarget { + sessionId: string; + refs: number; + destination?: BashAppendDestination; + pending?: Promise; +} + +interface PendingBashMessage { + target: BashSessionTarget; + message: BashExecutionMessage; +} + +/** Ownership snapshot spanning a session or branch transition. */ +export interface BashSessionTransition { + oldTarget: BashSessionTarget; + newTarget: BashSessionTarget; + oldSessionId: string; + oldSessionFile: string | undefined; + oldLeafId: string | null; + detachedManager: SessionManager | undefined; + resolveOld: ((destination: BashAppendDestination) => void) | undefined; + resolveNew: (destination: BashAppendDestination) => void; +} + +/** Capabilities the bash runner borrows from its owning session. */ +export interface BashRunnerHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + extensionRunner(): ExtensionRunner | undefined; + isStreaming(): boolean; +} + +/** Owns bash execution and preserves result ownership across transcript transitions. */ +export class BashRunner { + readonly #host: BashRunnerHost; + #abortControllers = new Set(); + #pendingMessages: PendingBashMessage[] = []; + #sessionTarget: BashSessionTarget; + + constructor(host: BashRunnerHost) { + this.#host = host; + this.#sessionTarget = { + sessionId: host.sessionManager.getSessionId(), + refs: 0, + destination: { kind: "current", manager: host.sessionManager }, + }; + } + + /** Executes a bash command while retaining the session and branch that owned its start. */ + async executeBash( + command: string, + onChunk?: (chunk: string) => void, + options?: { excludeFromContext?: boolean; useUserShell?: boolean }, + ): Promise { + const target = this.#captureSessionTarget(); + let targetTransferred = false; + const excludeFromContext = options?.excludeFromContext === true; + const cwd = this.#host.sessionManager.getCwd(); + try { + const extensionRunner = this.#host.extensionRunner(); + if (extensionRunner?.hasHandlers("user_bash")) { + const hookResult = await extensionRunner.emitUserBash({ + type: "user_bash", + command, + excludeFromContext, + cwd, + }); + if (hookResult?.result) { + targetTransferred = true; + await this.#recordResultForTarget(target, command, hookResult.result, options); + return hookResult.result; + } + } + + const abortController = new AbortController(); + this.#abortControllers.add(abortController); + let result: BashResult; + try { + result = await executeBashCommand(command, { + onChunk, + signal: abortController.signal, + sessionKey: target.sessionId, + cwd, + timeout: clampTimeout("bash", undefined, this.#host.settings.get("tools.maxTimeout")) * 1000, + onMinimizedSave: originalText => this.#saveOriginalArtifact(target, originalText), + useUserShell: options?.useUserShell, + }); + } finally { + this.#abortControllers.delete(abortController); + } + targetTransferred = true; + await this.#recordResultForTarget(target, command, result, options); + return result; + } finally { + if (!targetTransferred) await this.#releaseSessionTarget(target); + } + } + + /** Records a bash result supplied outside executeBash in the current ownership scope. */ + recordBashResult(command: string, result: BashResult, options?: { excludeFromContext?: boolean }): void { + const target = this.#captureSessionTarget(); + const message = this.#createMessage(command, result, options); + if (this.#host.isStreaming() && target === this.#sessionTarget) { + this.#pendingMessages.push({ target, message }); + return; + } + if (target.destination) { + try { + this.#appendMessage(target.destination, message); + } finally { + void this.#releaseSessionTarget(target); + } + return; + } + void this.#appendOwnedMessage(target, message).catch(error => { + logger.error("Failed to record bash result in its owning session", { error: String(error) }); + }); + } + + /** Cancels every running bash command. */ + abort(): void { + for (const abortController of this.#abortControllers) abortController.abort(); + } + + /** Whether a bash command is currently running. */ + get isRunning(): boolean { + return this.#abortControllers.size > 0; + } + + /** Whether bash results are waiting for a safe persistence boundary. */ + get hasPendingMessages(): boolean { + return this.#pendingMessages.length > 0; + } + + /** Flushes deferred bash results without changing their captured ownership. */ + async flushPending(): Promise { + if (this.#pendingMessages.length === 0) return; + const pending = this.#pendingMessages; + this.#pendingMessages = []; + for (const { target, message } of pending) await this.#appendOwnedMessage(target, message); + } + + /** Runs a leaf rewrite while retaining in-flight bash on its originating branch. */ + withBranchTransition(mutate: () => T): T { + const transition = this.beginSessionTransition(); + let transitioned = false; + try { + const result = mutate(); + this.markSessionTransition(transition); + transitioned = true; + return result; + } finally { + this.finishSessionTransition(transition, transitioned); + } + } + + /** Snapshots the owner of in-flight bash before a session or branch transition. */ + beginSessionTransition(options?: { persistDetached?: boolean }): BashSessionTransition { + const oldTarget = this.#sessionTarget; + let detachedManager: SessionManager | undefined; + let resolveOld: ((destination: BashAppendDestination) => void) | undefined; + if (oldTarget.refs > 0) { + detachedManager = this.#host.sessionManager.cloneCurrentSession({ persist: options?.persistDetached }); + const pendingOld = Promise.withResolvers(); + oldTarget.destination = undefined; + oldTarget.pending = pendingOld.promise; + resolveOld = pendingOld.resolve; + } + const pendingNew = Promise.withResolvers(); + return { + oldTarget, + newTarget: { + sessionId: this.#host.sessionManager.getSessionId(), + refs: 0, + pending: pendingNew.promise, + }, + oldSessionId: this.#host.sessionManager.getSessionId(), + oldSessionFile: this.#host.sessionManager.getSessionFile(), + oldLeafId: this.#host.sessionManager.getLeafId(), + detachedManager, + resolveOld, + resolveNew: pendingNew.resolve, + }; + } + + /** Adopts a transition's new target as the live bash owner. */ + markSessionTransition(transition: BashSessionTransition): void { + transition.newTarget.sessionId = this.#host.sessionManager.getSessionId(); + this.#sessionTarget = transition.newTarget; + } + + /** Resolves destinations opened by beginSessionTransition. */ + finishSessionTransition(transition: BashSessionTransition, success: boolean): void { + const manager = this.#host.sessionManager; + const currentDestination: BashAppendDestination = { kind: "current", manager }; + let oldDestination: BashAppendDestination = currentDestination; + if (success && transition.resolveOld) { + const currentFile = manager.getSessionFile(); + const sameFile = + transition.oldSessionFile === currentFile || + (transition.oldSessionFile !== undefined && + currentFile !== undefined && + path.resolve(transition.oldSessionFile) === path.resolve(currentFile)); + const sameSession = transition.oldSessionId === manager.getSessionId() && sameFile; + if (sameSession) { + oldDestination = + transition.oldLeafId === manager.getLeafId() + ? currentDestination + : { kind: "branch", manager, parentId: transition.oldLeafId }; + } else if (transition.detachedManager) { + oldDestination = { kind: "detached", manager: transition.detachedManager }; + } + } + if (transition.resolveOld) { + transition.oldTarget.pending = undefined; + transition.oldTarget.destination = oldDestination; + transition.resolveOld(oldDestination); + } + transition.newTarget.pending = undefined; + transition.newTarget.destination = currentDestination; + if (!success) transition.newTarget.sessionId = manager.getSessionId(); + transition.resolveNew(currentDestination); + if (transition.detachedManager && (oldDestination.kind !== "detached" || transition.oldTarget.refs === 0)) { + void transition.detachedManager.close().catch(error => { + logger.warn("Failed to close detached bash session writer", { error: String(error) }); + }); + } + } + + async #saveOriginalArtifact(target: BashSessionTarget, originalText: string): Promise { + try { + const destination = target.destination ?? (await target.pending); + return await destination?.manager.saveArtifact(originalText, "bash-original"); + } catch { + return undefined; + } + } + + #createMessage( + command: string, + result: BashResult, + options?: { excludeFromContext?: boolean }, + ): BashExecutionMessage { + const meta = outputMeta().truncationFromSummary(result, { direction: "tail" }).get(); + return { + role: "bashExecution", + command, + output: result.output, + exitCode: result.exitCode, + cancelled: result.cancelled, + truncated: result.truncated, + meta, + timestamp: Date.now(), + excludeFromContext: options?.excludeFromContext, + }; + } + + #captureSessionTarget(): BashSessionTarget { + this.#sessionTarget.refs++; + return this.#sessionTarget; + } + + async #releaseSessionTarget(target: BashSessionTarget): Promise { + if (target.refs <= 0) throw new Error("Bash session target released more than once"); + target.refs--; + if (target.refs === 0 && target.destination?.kind === "detached") await target.destination.manager.close(); + } + + #appendMessage(destination: BashAppendDestination, message: BashExecutionMessage): void { + switch (destination.kind) { + case "current": + this.#host.agent.appendMessage(message); + destination.manager.appendMessage(message); + break; + case "detached": + destination.manager.appendMessage(message); + break; + case "branch": + destination.parentId = destination.manager.appendMessageToBranch(message, destination.parentId); + break; + } + } + + async #appendOwnedMessage(target: BashSessionTarget, message: BashExecutionMessage): Promise { + try { + const destination = target.destination ?? (await target.pending); + if (!destination) throw new Error("Bash session target has no append destination"); + this.#appendMessage(destination, message); + } finally { + await this.#releaseSessionTarget(target); + } + } + + async #recordResultForTarget( + target: BashSessionTarget, + command: string, + result: BashResult, + options?: { excludeFromContext?: boolean }, + ): Promise { + const message = this.#createMessage(command, result, options); + if (this.#host.isStreaming() && target === this.#sessionTarget) { + this.#pendingMessages.push({ target, message }); + return; + } + await this.#appendOwnedMessage(target, message); + } +} diff --git a/packages/coding-agent/src/session/checkpoint-entries.ts b/packages/coding-agent/src/session/checkpoint-entries.ts new file mode 100644 index 000000000..61b0b501e --- /dev/null +++ b/packages/coding-agent/src/session/checkpoint-entries.ts @@ -0,0 +1,81 @@ +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; +import { stringProperty } from "@oh-my-pi/pi-utils"; +import type { CompletedRewindState } from "../tools/checkpoint"; +import { writeDeviceDispatch } from "../tools/resolve"; +import type { SessionEntry } from "./session-entries"; + +/** Extracts text from custom message content. */ +export function customMessageContentText(content: string | (TextContent | ImageContent)[]): string { + if (typeof content === "string") return content; + const parts: string[] = []; + for (const part of content) { + if (part.type === "text") parts.push(part.text); + } + return parts.join("\n"); +} + +/** Extracts the report body from persisted rewind-report content. */ +export function reportFromRewindReportContent(content: string): string { + const marker = "\nReport:\n"; + const index = content.lastIndexOf(marker); + const report = index >= 0 ? content.slice(index + marker.length) : content; + return report.trim(); +} + +/** Checkpoint-domain tool names normalized from native and xdev calls. */ +export type SemanticCheckpointToolName = "checkpoint" | "rewind"; + +/** Normalized checkpoint-domain tool result. */ +export interface SemanticToolResult { + toolName: SemanticCheckpointToolName; + details?: unknown; +} + +/** Normalizes checkpoint and rewind results across native calls and xdev dispatches. */ +export function semanticToolResult(toolName: string | undefined, result: unknown): SemanticToolResult | undefined { + if (toolName === "checkpoint" || toolName === "rewind") { + const details = result && typeof result === "object" && "details" in result ? result.details : undefined; + return { toolName, details }; + } + const dispatch = writeDeviceDispatch(toolName ?? "", result); + if (dispatch?.mode !== "execute" || (dispatch.tool !== "checkpoint" && dispatch.tool !== "rewind")) { + return undefined; + } + return { toolName: dispatch.tool, details: dispatch.inner }; +} + +/** Restores completed rewind state from a persisted session entry. */ +export function completedRewindFromEntry(entry: SessionEntry): CompletedRewindState | undefined { + if (entry.type !== "custom_message" || entry.customType !== "rewind-report") return undefined; + const details = entry.details; + if (!details || typeof details !== "object") return undefined; + const startedAt = stringProperty(details, "startedAt"); + const rewoundAt = stringProperty(details, "rewoundAt"); + if (!startedAt || !rewoundAt) return undefined; + const report = + stringProperty(details, "report")?.trim() || + reportFromRewindReportContent(customMessageContentText(entry.content)); + return report.length > 0 ? { report, startedAt, rewoundAt } : undefined; +} + +/** Whether an entry is a successful checkpoint tool result. */ +export function isSuccessfulCheckpointEntry( + entry: SessionEntry, +): entry is SessionEntry & { type: "message"; message: Extract } { + if (entry.type !== "message" || entry.message.role !== "toolResult" || entry.message.isError === true) { + return false; + } + return semanticToolResult(entry.message.toolName, entry.message)?.toolName === "checkpoint"; +} + +/** Returns the checkpoint start timestamp represented by an entry. */ +export function checkpointStartedAtFromEntry(entry: SessionEntry): string | undefined { + if (!isSuccessfulCheckpointEntry(entry)) return undefined; + const details = semanticToolResult(entry.message.toolName, entry.message)?.details; + if (details && typeof details === "object") { + const startedAt = stringProperty(details, "startedAt"); + if (startedAt) return startedAt; + } + return entry.timestamp; +} diff --git a/packages/coding-agent/src/session/eval-runner.ts b/packages/coding-agent/src/session/eval-runner.ts new file mode 100644 index 000000000..634cf09d4 --- /dev/null +++ b/packages/coding-agent/src/session/eval-runner.ts @@ -0,0 +1,212 @@ +import type { Agent } from "@oh-my-pi/pi-agent-core"; +import { logger } from "@oh-my-pi/pi-utils"; +import type { Settings } from "../config/settings"; +import { disposeJuliaKernelSessionsByOwner } from "../eval/jl/executor"; +import { namespaceSessionId as namespacePythonSessionId } from "../eval/py"; +import { + disposeKernelSessionsByOwner, + executePython as executePythonCommand, + type PythonResult, +} from "../eval/py/executor"; +import { disposeRubyKernelSessionsByOwner } from "../eval/rb/executor"; +import { defaultEvalSessionId } from "../eval/session-id"; +import type { ExtensionRunner } from "../extensibility/extensions"; +import { outputMeta } from "../tools/output-meta"; +import type { PythonExecutionMessage } from "./messages"; +import type { SessionManager } from "./session-manager"; + +/** Capabilities the eval runner borrows from its owning session. */ +export interface EvalRunnerHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + extensionRunner(): ExtensionRunner | undefined; + isStreaming(): boolean; + appendSessionMessage(message: PythonExecutionMessage): void; +} + +/** Owns user-initiated Python execution and retained eval-kernel lifecycle. */ +export class EvalRunner { + readonly #host: EvalRunnerHost; + readonly #kernelOwnerId: string; + readonly #parentSessionId: string | undefined; + #abortControllers = new Set(); + #pendingMessages: PythonExecutionMessage[] = []; + #activeExecutions = new Set>(); + #disposing = false; + + constructor(host: EvalRunnerHost, options: { kernelOwnerId: string; parentSessionId: string | undefined }) { + this.#host = host; + this.#kernelOwnerId = options.kernelOwnerId; + this.#parentSessionId = options.parentSessionId; + } + + /** Executes Python in the session's shared kernel. */ + async executePython( + code: string, + onChunk?: (chunk: string) => void, + options?: { excludeFromContext?: boolean }, + ): Promise { + const excludeFromContext = options?.excludeFromContext === true; + const cwd = this.#host.sessionManager.getCwd(); + this.assertExecutionAllowed(); + const abortController = new AbortController(); + const execution = (async (): Promise => { + const extensionRunner = this.#host.extensionRunner(); + if (extensionRunner?.hasHandlers("user_python")) { + const hookResult = await extensionRunner.emitUserPython({ + type: "user_python", + code, + excludeFromContext, + cwd, + }); + this.assertExecutionAllowed(); + if (hookResult?.result) { + this.recordPythonResult(code, hookResult.result, options); + return hookResult.result; + } + } + const sessionId = + this.getSessionId() ?? + defaultEvalSessionId({ + cwd, + getSessionFile: () => this.#host.sessionManager.getSessionFile() ?? null, + }); + const result = await executePythonCommand(code, { + cwd, + sessionId: namespacePythonSessionId(sessionId), + kernelOwnerId: this.#kernelOwnerId, + kernelMode: this.#host.settings.get("python.kernelMode"), + interpreter: this.#host.settings.get("python.interpreter")?.trim() || undefined, + onChunk, + signal: abortController.signal, + }); + this.recordPythonResult(code, result, options); + return result; + })(); + return await this.trackExecution(execution, abortController); + } + + /** Rejects new eval work once session disposal begins. */ + assertExecutionAllowed(): void { + if (this.#disposing) throw new Error("Python execution is unavailable while session disposal is in progress"); + } + + /** Tracks externally started Python work so disposal can await and abort it. */ + trackExecution(execution: Promise, abortController: AbortController): Promise { + this.#abortControllers.add(abortController); + this.#activeExecutions.add(execution); + void execution.then( + () => { + this.#abortControllers.delete(abortController); + this.#activeExecutions.delete(execution); + }, + () => { + this.#abortControllers.delete(abortController); + this.#activeExecutions.delete(execution); + }, + ); + return execution; + } + + /** Records a Python execution result in session history. */ + recordPythonResult(code: string, result: PythonResult, options?: { excludeFromContext?: boolean }): void { + const meta = outputMeta().truncationFromSummary(result, { direction: "tail" }).get(); + const message: PythonExecutionMessage = { + role: "pythonExecution", + code, + output: result.output, + exitCode: result.exitCode, + cancelled: result.cancelled, + truncated: result.truncated, + meta, + timestamp: Date.now(), + excludeFromContext: options?.excludeFromContext, + }; + if (this.#host.isStreaming()) { + this.#pendingMessages.push(message); + } else { + this.#host.appendSessionMessage(message); + } + } + + /** Cancels every running Python execution. */ + abort(): void { + for (const abortController of this.#abortControllers) abortController.abort(); + } + + /** Whether a Python execution is currently running. */ + get isRunning(): boolean { + return this.#abortControllers.size > 0; + } + + /** Whether Python results are waiting for a safe persistence boundary. */ + get hasPendingMessages(): boolean { + return this.#pendingMessages.length > 0; + } + + /** Returns the eval session shared with the Python backend. */ + getSessionId(): string | null { + if (this.#parentSessionId !== undefined) return this.#parentSessionId; + return defaultEvalSessionId({ + cwd: this.#host.sessionManager.getCwd(), + getSessionFile: () => this.#host.sessionManager.getSessionFile() ?? null, + }); + } + + /** Flushes deferred Python results into agent state and persistence. */ + flushPending(): void { + if (this.#pendingMessages.length === 0) return; + for (const message of this.#pendingMessages) this.#host.appendSessionMessage(message); + this.#pendingMessages = []; + } + + /** Prevents new Python executions before asynchronous disposal starts. */ + beginDispose(): void { + this.#disposing = true; + } + + /** Waits for active work and disposes every retained eval kernel owned by the session. */ + async disposeKernels(): Promise { + const settled = await this.#prepareExecutionsForDispose(); + if (!settled) { + logger.warn("Detaching retained eval-kernel ownership during dispose while eval execution is still active"); + } + const results = await Promise.allSettled([ + disposeKernelSessionsByOwner(this.#kernelOwnerId), + disposeRubyKernelSessionsByOwner(this.#kernelOwnerId), + disposeJuliaKernelSessionsByOwner(this.#kernelOwnerId), + ]); + const errors: unknown[] = []; + for (const result of results) if (result.status === "rejected") errors.push(result.reason); + if (errors.length > 0) throw new AggregateError(errors, "Failed to dispose one or more eval kernels"); + } + + async #waitForExecutionsToSettle(timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs; + while (this.#activeExecutions.size > 0) { + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) return false; + const settled = await Promise.race([ + Promise.allSettled(Array.from(this.#activeExecutions)).then(() => true), + Bun.sleep(remainingMs).then(() => false), + ]); + if (!settled && this.#activeExecutions.size > 0) return false; + } + return true; + } + + async #prepareExecutionsForDispose(): Promise { + if (!(await this.#waitForExecutionsToSettle(3_000))) { + logger.warn("Aborting active Python execution during dispose before retained kernel cleanup"); + this.abort(); + if (!(await this.#waitForExecutionsToSettle(1_000))) { + logger.warn( + "Python execution is still active after dispose aborted all active runs; retained kernel ownership will still be detached", + ); + return false; + } + } + return true; + } +} diff --git a/packages/coding-agent/src/session/irc-bridge.ts b/packages/coding-agent/src/session/irc-bridge.ts new file mode 100644 index 000000000..831fa2e76 --- /dev/null +++ b/packages/coding-agent/src/session/irc-bridge.ts @@ -0,0 +1,203 @@ +import type { Agent } from "@oh-my-pi/pi-agent-core"; +import { logger, prompt } from "@oh-my-pi/pi-utils"; +import type { Settings } from "../config/settings"; +import { IrcBus, type IrcMessage } from "../irc/bus"; +import parentIrcSteerTemplate from "../prompts/steering/parent-irc.md" with { type: "text" }; +import ircAutoReplyTemplate from "../prompts/system/irc-autoreply.md" with { type: "text" }; +import ircIncomingTemplate from "../prompts/system/irc-incoming.md" with { type: "text" }; +import { AgentRegistry } from "../registry/agent-registry"; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { CustomMessage } from "./messages"; +import type { SessionManager } from "./session-manager"; + +/** Capabilities the IRC bridge borrows from its owning session. */ +export interface IrcBridgeHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + isDisposed(): boolean; + isStreaming(): boolean; + planModeEnabled(): boolean; + emitSessionEvent(event: AgentSessionEvent): Promise; + wakeForIrc(records: CustomMessage[]): void; + runEphemeralTurn(args: { promptText: string }): Promise<{ replyText: string }>; +} + +/** Owns incoming IRC queues, injection, and side-channel auto-replies. */ +export class IrcBridge { + readonly #host: IrcBridgeHost; + #interrupts: CustomMessage[] = []; + #asides: CustomMessage[] = []; + + constructor(host: IrcBridgeHost) { + this.#host = host; + } + + /** Whether an incoming peer message can interrupt a wait. */ + hasInterrupts(): boolean { + return this.#interrupts.length > 0; + } + + /** Whether any undelivered IRC record remains queued. */ + hasPending(): boolean { + return this.#interrupts.length > 0 || this.#asides.length > 0; + } + + /** Takes every queued IRC record in interrupt-before-aside order. */ + drainPending(): CustomMessage[] { + const records = [...this.#interrupts, ...this.#asides]; + this.#interrupts = []; + this.#asides = []; + return records; + } + + /** Surfaces and consumes queued incoming records before automatic injection. */ + drainInboxMessages(agentId: string, opts?: { from?: string; limit?: number }): IrcMessage[] { + const messages: IrcMessage[] = []; + const remainingInterrupts: CustomMessage[] = []; + const remainingAsides: CustomMessage[] = []; + const queues = [ + { records: this.#interrupts, remaining: remainingInterrupts }, + { records: this.#asides, remaining: remainingAsides }, + ]; + for (const queue of queues) { + for (const record of queue.records) { + if (record.customType !== "irc:incoming") { + queue.remaining.push(record); + continue; + } + const details = record.details; + if (!details || typeof details !== "object") { + queue.remaining.push(record); + continue; + } + const id = Reflect.get(details, "id"); + const from = Reflect.get(details, "from"); + const body = Reflect.get(details, "message"); + const replyTo = Reflect.get(details, "replyTo"); + if (typeof id !== "string" || typeof from !== "string" || typeof body !== "string") { + queue.remaining.push(record); + continue; + } + if (opts?.from !== undefined && from !== opts.from) { + queue.remaining.push(record); + continue; + } + if (opts?.limit !== undefined && messages.length >= opts.limit) { + queue.remaining.push(record); + continue; + } + messages.push({ + id, + from, + to: agentId, + body, + ts: record.timestamp, + ...(typeof replyTo === "string" ? { replyTo } : {}), + }); + } + } + this.#interrupts = remainingInterrupts; + this.#asides = remainingAsides; + return messages; + } + + /** Delivers an IRC message into the recipient session without awaiting any wake turn. */ + async deliver(msg: IrcMessage, opts?: { expectsReply?: boolean }): Promise<"injected" | "woken"> { + if (this.#host.isDisposed()) throw new Error("Recipient session is disposed."); + const streaming = this.#host.isStreaming(); + const planModeIdle = !streaming && this.#host.planModeEnabled(); + const autoReply = + (opts?.expectsReply ?? false) && ((streaming && !this.#host.settings.get("async.enabled")) || planModeIdle); + const record: CustomMessage = { + role: "custom", + customType: "irc:incoming", + content: prompt.render(ircIncomingTemplate, { + from: msg.from, + message: msg.body, + replyTo: msg.replyTo ?? "", + autoReplied: autoReply, + interrupting: streaming, + }), + display: true, + details: { id: msg.id, from: msg.from, message: msg.body, ...(msg.replyTo ? { replyTo: msg.replyTo } : {}) }, + attribution: "agent", + timestamp: msg.ts, + }; + void this.#host.emitSessionEvent({ type: "irc_message", message: record }); + if (streaming) { + const recipientParentId = AgentRegistry.global().get(msg.to)?.parentId; + if (recipientParentId === msg.from) { + this.#host.agent.steer({ + role: "user", + content: prompt.render(parentIrcSteerTemplate, { from: msg.from, message: msg.body }), + attribution: "agent", + timestamp: msg.ts, + steering: true, + }); + } else { + this.#interrupts.push(record); + } + if (autoReply) void this.#runAutoReply(msg); + return "injected"; + } + if (this.#host.planModeEnabled()) { + this.#host.agent.appendMessage(record); + this.#host.sessionManager.appendCustomMessageEntry( + record.customType, + record.content, + record.display, + record.details, + record.attribution ?? "agent", + ); + if (autoReply) void this.#runAutoReply(msg); + return "injected"; + } + this.#host.wakeForIrc([record]); + return "woken"; + } + + /** Emits an IRC relay observation for rendering without persisting it. */ + emitRelayObservation(record: CustomMessage): void { + void this.#host.emitSessionEvent({ type: "irc_message", message: record }); + } + + /** Persists queued IRC records that missed their step-boundary injection. */ + flushPending(): void { + for (const record of this.drainPending()) { + this.#host.agent.emitExternalEvent({ type: "message_start", message: record }); + this.#host.agent.emitExternalEvent({ type: "message_end", message: record }); + } + } + + async #runAutoReply(msg: IrcMessage): Promise { + try { + const { replyText } = await this.#host.runEphemeralTurn({ + promptText: prompt.render(ircAutoReplyTemplate, { + from: msg.from, + message: msg.body, + replyTo: msg.replyTo ?? "", + }), + }); + const body = replyText.trim(); + if (!body || this.#host.isDisposed()) return; + const record: CustomMessage = { + role: "custom", + customType: "irc:autoreply", + content: `[IRC you → \`${msg.from}\` (auto)]\n\n${body}`, + display: true, + details: { to: msg.from, body, replyTo: msg.id }, + attribution: "agent", + timestamp: Date.now(), + }; + void this.#host.emitSessionEvent({ type: "irc_message", message: record }); + this.#asides.push(record); + const receipt = await IrcBus.global().send({ from: msg.to, to: msg.from, body, replyTo: msg.id }); + if (receipt.outcome === "failed") { + logger.warn("IRC auto-reply delivery failed", { to: msg.from, error: receipt.error }); + } + } catch (error) { + logger.warn("IRC auto-reply turn failed", { from: msg.from, error: String(error) }); + } + } +} diff --git a/packages/coding-agent/src/session/messages.ts b/packages/coding-agent/src/session/messages.ts index 97a607ddf..dbc868fc2 100644 --- a/packages/coding-agent/src/session/messages.ts +++ b/packages/coding-agent/src/session/messages.ts @@ -23,8 +23,9 @@ import type { UserMessage, } from "@oh-my-pi/pi-ai"; import * as AIError from "@oh-my-pi/pi-ai/error"; -import { prompt } from "@oh-my-pi/pi-utils"; +import { isRecord, logger, prompt } from "@oh-my-pi/pi-utils"; import userInterjectionTemplate from "../prompts/steering/user-interjection.md" with { type: "text" }; +import { formatTitleConversationContext, type TitleConversationTurn } from "../tiny/message-preproc"; export { type BranchSummaryMessage, @@ -41,6 +42,267 @@ export const SKILL_PROMPT_MESSAGE_TYPE = "skill-prompt"; export const LSP_LATE_DIAGNOSTIC_MESSAGE_TYPE = "lsp-late-diagnostic"; export const BACKGROUND_TAN_DISPATCH_MESSAGE_TYPE = "background-tan-dispatch"; +/** + * Logs provider-error turns so their actual cause is available outside the + * session transcript. No-op for non-error stop reasons. + */ +export function logProviderTurnError(msg: AssistantMessage): void { + if (msg.stopReason !== "error") return; + logger.warn("agent turn ended with provider error", { + provider: msg.provider, + model: msg.model, + errorMessage: msg.errorMessage, + errorStatus: msg.errorStatus, + errorId: msg.errorId, + }); +} + +const EPHEMERAL_REPLY_MAX_BYTES = 4096; +const REPLAN_TITLE_CONTEXT_TURN_LIMIT = 6; + +/** + * Removes replay-bound provider state before reparenting an assistant message + * under a different user turn. + */ +export function sanitizeAssistantForReparentedHistory(message: AssistantMessage): AssistantMessage { + const content: AssistantMessage["content"] = []; + for (const block of message.content) { + if (block.type === "redactedThinking") continue; + if (block.type === "thinking") { + content.push({ type: "thinking", thinking: block.thinking }); + continue; + } + content.push(block); + } + return { ...message, content, providerPayload: undefined }; +} + +/** + * Collapses degenerate repeated lines and bounds an ephemeral side-channel + * reply to 4 KiB. + */ +export function dedupeEphemeralReply(text: string): string { + if (!text) return text; + const lines = text.split("\n"); + const out: string[] = []; + let i = 0; + while (i < lines.length) { + let j = i + 1; + while (j < lines.length && lines[j] === lines[i]) j++; + const runLen = j - i; + if (runLen > 3) { + out.push(lines[i], `[…${runLen}×]`); + } else { + for (let k = 0; k < runLen; k++) out.push(lines[i]); + } + i = j; + } + let result = out.join("\n"); + if (Buffer.byteLength(result, "utf8") > EPHEMERAL_REPLY_MAX_BYTES) { + const suffix = "\n[…truncated]"; + const budget = EPHEMERAL_REPLY_MAX_BYTES - Buffer.byteLength(suffix, "utf8"); + while (Buffer.byteLength(result, "utf8") > budget) { + result = result.slice(0, -1); + } + result += suffix; + } + return result; +} + +/** Builds the recent user/assistant context supplied to title regeneration. */ +export function buildReplanTitleContext(messages: AgentMessage[]): string { + const turns: TitleConversationTurn[] = []; + for (let i = messages.length - 1; i >= 0 && turns.length < REPLAN_TITLE_CONTEXT_TURN_LIMIT; i--) { + const message = messages[i]; + if (!message) continue; + const turn = titleConversationTurnFromMessage(message); + if (turn) turns.push(turn); + } + turns.reverse(); + return formatTitleConversationContext(turns); +} + +/** + * Compares session messages by provider-replay semantics, ignoring runtime-only + * fields that do not change a restored request. + */ +export function didSessionMessagesChange(previousMessages: AgentMessage[], nextMessages: AgentMessage[]): boolean { + if (previousMessages.length !== nextMessages.length) return true; + return previousMessages.some( + (message, i) => + !Bun.deepEquals( + normalizeSessionMessageForProviderReplay(message), + normalizeSessionMessageForProviderReplay(nextMessages[i]), + ), + ); +} + +function textFromContent(content: unknown): string { + if (typeof content === "string") return content.trim(); + if (!Array.isArray(content)) return ""; + const parts: string[] = []; + for (const block of content) { + if (!isRecord(block) || block.type !== "text" || typeof block.text !== "string") continue; + const text = block.text.trim(); + if (text) parts.push(text); + } + return parts.join("\n\n"); +} + +function thinkingFromContent(content: unknown): string { + if (!Array.isArray(content)) return ""; + const parts: string[] = []; + for (const block of content) { + if (!isRecord(block) || block.type !== "thinking" || typeof block.thinking !== "string") continue; + const thinking = block.thinking.trim(); + if (thinking) parts.push(thinking); + } + return parts.join("\n\n"); +} + +function titleConversationTurnFromMessage(message: AgentMessage): TitleConversationTurn | undefined { + if (message.role !== "user" && message.role !== "assistant") return undefined; + const text = textFromContent(message.content); + const thinking = message.role === "assistant" ? thinkingFromContent(message.content) : undefined; + if (!text && !thinking) return undefined; + return { role: message.role, ...(text ? { text } : {}), ...(thinking ? { thinking } : {}) }; +} + +function normalizeProviderReplayValue(value: unknown): unknown { + if (Array.isArray(value)) { + return value.map(normalizeProviderReplayValue); + } + if (value && typeof value === "object") { + return Object.fromEntries( + Object.entries(value).map(([key, entryValue]) => [key, normalizeProviderReplayValue(entryValue)]), + ); + } + return value; +} + +function normalizeSessionMessageForProviderReplay(message: AgentMessage): unknown { + switch (message.role) { + case "user": + case "developer": + return { + role: message.role, + content: normalizeProviderReplayValue(message.content), + providerPayload: message.providerPayload, + }; + case "assistant": { + const isResponsesFamilyMessage = + message.api === "openai-responses" || message.api === "openai-codex-responses"; + return { + role: message.role, + content: + isResponsesFamilyMessage && Array.isArray(message.content) + ? message.content.flatMap(block => { + if (block.type === "thinking") { + return []; + } + if (block.type === "toolCall") { + return [ + { + type: block.type, + id: block.id, + name: block.name, + arguments: block.arguments, + }, + ]; + } + if (block.type === "text") { + return [{ type: block.type, text: block.text, textSignature: block.textSignature }]; + } + return [normalizeProviderReplayValue(block)]; + }) + : normalizeProviderReplayValue(message.content), + api: message.api, + provider: message.provider, + model: message.model, + stopReason: message.stopReason, + errorMessage: message.errorMessage, + providerPayload: isResponsesFamilyMessage ? undefined : message.providerPayload, + }; + } + case "toolResult": + return { + role: message.role, + toolName: message.toolName, + toolCallId: message.toolCallId, + isError: message.isError, + content: normalizeProviderReplayValue(message.content), + }; + case "bashExecution": + return { + role: message.role, + command: message.command, + output: message.output, + exitCode: message.exitCode, + cancelled: message.cancelled, + meta: message.meta + ? { + truncation: normalizeProviderReplayValue(message.meta.truncation), + limits: normalizeProviderReplayValue(message.meta.limits), + diagnostics: message.meta.diagnostics + ? normalizeProviderReplayValue({ + summary: message.meta.diagnostics.summary, + messages: message.meta.diagnostics.messages, + }) + : undefined, + } + : undefined, + excludeFromContext: message.excludeFromContext, + }; + case "pythonExecution": + return { + role: message.role, + code: message.code, + output: message.output, + exitCode: message.exitCode, + cancelled: message.cancelled, + meta: message.meta + ? { + truncation: normalizeProviderReplayValue(message.meta.truncation), + limits: normalizeProviderReplayValue(message.meta.limits), + diagnostics: message.meta.diagnostics + ? normalizeProviderReplayValue({ + summary: message.meta.diagnostics.summary, + messages: message.meta.diagnostics.messages, + }) + : undefined, + } + : undefined, + excludeFromContext: message.excludeFromContext, + }; + case "custom": + case "hookMessage": + return { + role: message.role, + customType: message.customType, + content: normalizeProviderReplayValue(message.content), + }; + case "branchSummary": + return { role: message.role, summary: message.summary }; + case "compactionSummary": + return { + role: message.role, + summary: message.summary, + providerPayload: message.providerPayload, + }; + case "fileMention": + return { + role: message.role, + files: message.files.map(file => ({ + path: file.path, + content: file.content, + image: file.image, + })), + }; + default: + return normalizeProviderReplayValue(message); + } +} + /** Fallback type for extension-injected messages that omit a custom type. */ export const DEFAULT_CUSTOM_MESSAGE_TYPE = "custom-message"; diff --git a/packages/coding-agent/src/session/model-controls.ts b/packages/coding-agent/src/session/model-controls.ts new file mode 100644 index 000000000..f03ef0f30 --- /dev/null +++ b/packages/coding-agent/src/session/model-controls.ts @@ -0,0 +1,714 @@ +import { type Agent, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import type { Model, ProviderSessionState, ServiceTier, ServiceTierByFamily, ServiceTierFamily } from "@oh-my-pi/pi-ai"; +import { + clearAnthropicFastModeFallback, + Effort, + realizesPriorityServiceTier, + resolveModelServiceTier, + serviceTierFamily, +} from "@oh-my-pi/pi-ai"; +import { isFireworksFastModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; +import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; +import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; +import { logger } from "@oh-my-pi/pi-utils"; +import { classifyDifficulty } from "../auto-thinking/classifier"; +import type { ModelRegistry } from "../config/model-registry"; +import { + filterAvailableModelsByEnabledPatterns, + formatModelStringWithRouting, + getModelMatchPreferences, + type ResolvedModelRoleValue, + resolveModelRoleValue, +} from "../config/model-resolver"; +import { getKnownRoleIds } from "../config/model-roles"; +import type { Settings } from "../config/settings"; +import { containsUltrathink } from "../modes/ultrathink"; +import { + AUTO_THINKING, + type ConfiguredThinkingLevel, + clampAutoThinkingEffort, + resolveProvisionalAutoLevel, + resolveThinkingLevelForModel, + shouldDisableReasoning, + toReasoningEffort, +} from "../thinking"; +import type { EditMode } from "../utils/edit-mode"; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { ModelCycleResult, ResolvedRoleModel, RoleModelCycle, RoleModelCycleResult } from "./agent-session-types"; +import { formatRoleModelValue, resolveRoleModelFull } from "./role-models"; +import { EPHEMERAL_MODEL_CHANGE_ROLE } from "./session-entries"; +import type { SessionManager } from "./session-manager"; + +/** Capabilities borrowed from the owning AgentSession. */ +export interface ModelControlsHost { + agent: Agent; + settings: Settings; + modelRegistry: ModelRegistry; + sessionManager: SessionManager; + providerSessionState: Map; + model(): Model | undefined; + sessionId(): string; + promptGeneration(): number; + resolveActiveEditMode(): EditMode; + syncAfterModelChange(previousEditMode: EditMode): Promise; + setModelWithProviderSessionReset(model: Model): void; + clearActiveRetryFallback(): void; + clearInheritedProviderPromptCacheKey(): void; + magicKeywordEnabled(keyword: "orchestrate" | "ultrathink" | "workflow"): boolean; + emit(event: AgentSessionEvent): void; + emitSessionEvent(event: AgentSessionEvent): Promise; + emitNotice(level: "info" | "warning" | "error", message: string, source?: string): void; +} + +/** Owns model selection, thinking effort, role cycling, and service tiers. */ +export class ModelControls { + readonly #host: ModelControlsHost; + #scopedModels: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; + #thinkingLevel: ThinkingLevel | undefined; + #autoThinking = false; + #autoResolvedLevel: Effort | undefined; + #serviceTierByFamily: ServiceTierByFamily; + + constructor( + host: ModelControlsHost, + options: { + scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; + thinkingLevel?: ConfiguredThinkingLevel; + serviceTierByFamily?: ServiceTierByFamily; + }, + ) { + this.#host = host; + this.#scopedModels = options.scopedModels ?? []; + this.#serviceTierByFamily = options.serviceTierByFamily ?? {}; + if (options.thinkingLevel === AUTO_THINKING) { + // Keep auto pending until the first turn while exposing a valid wire effort. + this.#autoThinking = true; + this.#thinkingLevel = resolveProvisionalAutoLevel(this.#model); + } else { + this.#thinkingLevel = options.thinkingLevel; + } + this.#applyThinkingLevelToAgent(this.#thinkingLevel); + } + + get #model(): Model | undefined { + return this.#host.model(); + } + + /** Effective metadata-clamped thinking level applied to the agent. */ + get thinkingLevel(): ThinkingLevel | undefined { + return this.#thinkingLevel; + } + + /** Configured selector, preserving `auto` while classification is active. */ + configuredThinkingLevel(): ConfiguredThinkingLevel | undefined { + return this.#autoThinking ? AUTO_THINKING : this.#thinkingLevel; + } + + /** Whether per-turn automatic thinking classification is enabled. */ + get isAutoThinking(): boolean { + return this.#autoThinking; + } + + /** Last concrete effort selected by automatic classification. */ + get autoResolvedThinkingLevel(): Effort | undefined { + return this.#autoResolvedLevel; + } + + /** Models explicitly scoped to the session's cycle command. */ + get scopedModels(): ReadonlyArray<{ model: Model; thinkingLevel?: ThinkingLevel }> { + return this.#scopedModels; + } + + /** Live per-provider-family service-tier selection. */ + get serviceTierByFamily(): ServiceTierByFamily { + return this.#serviceTierByFamily; + } + + /** Restores thinking state from a transcript without persisting a new entry. */ + restoreThinkingLevel(level: ConfiguredThinkingLevel | undefined): void { + this.#autoThinking = level === AUTO_THINKING; + this.#autoResolvedLevel = undefined; + this.#thinkingLevel = + level === AUTO_THINKING + ? resolveProvisionalAutoLevel(this.#model) + : resolveThinkingLevelForModel(this.#model, level); + this.#applyThinkingLevelToAgent(this.#thinkingLevel); + } + + /** Restores an exact thinking snapshot after a failed session switch. */ + restoreThinkingSnapshot(level: ThinkingLevel | undefined, auto: boolean, resolved: Effort | undefined): void { + this.#thinkingLevel = level; + this.#autoThinking = auto; + this.#autoResolvedLevel = resolved; + this.#applyThinkingLevelToAgent(level); + } + + /** Restores service tiers without persisting a duplicate transcript entry. */ + restoreServiceTiers(tiers: ServiceTierByFamily): void { + this.#serviceTierByFamily = tiers; + } + resolveRoleModel(role: string): Model | undefined { + return resolveRoleModelFull(this.#host.settings, role, this.#host.modelRegistry.getAvailable(), this.#model) + .model; + } + + resolveRoleModelWithThinking(role: string): ResolvedModelRoleValue { + return resolveRoleModelFull(this.#host.settings, role, this.#host.modelRegistry.getAvailable(), this.#model); + } + + resolveTemporaryModelThinkingLevel(model: Model): ConfiguredThinkingLevel | undefined { + const availableModels = this.#host.modelRegistry.getAvailable(); + if (availableModels.length === 0) return undefined; + + const matchPreferences = getModelMatchPreferences(this.#host.settings); + for (const role of getKnownRoleIds(this.#host.settings)) { + const roleValue = this.#host.settings.getModelRole(role); + if (!roleValue) continue; + + const resolved = resolveModelRoleValue(roleValue, availableModels, { + settings: this.#host.settings, + matchPreferences, + }); + if (!resolved.explicitThinkingLevel || resolved.thinkingLevel === undefined || !resolved.model) continue; + if (modelsAreEqual(resolved.model, model)) return resolved.thinkingLevel; + } + + return undefined; + } + + async setModel( + model: Model, + role: string = "default", + options?: { + selector?: string; + thinkingLevel?: ThinkingLevel; + persist?: boolean; + currentContextTokens?: number; + }, + ): Promise<{ switched: boolean }> { + const previousEditMode = this.#host.resolveActiveEditMode(); + if (!this.#host.modelRegistry.hasConfiguredAuth(model)) { + throw new Error(`No API key for ${model.provider}/${model.id}`); + } + + const targetModel = await this.#host.modelRegistry.refreshSelectedModelMetadata(model); + + this.#host.modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); + this.#host.clearActiveRetryFallback(); + this.#host.setModelWithProviderSessionReset(targetModel); + this.#host.sessionManager.appendModelChange(`${targetModel.provider}/${targetModel.id}`, role); + if (options?.persist) { + this.#host.settings.setModelRole( + role, + formatRoleModelValue( + this.#host.settings, + this.#host.modelRegistry, + role, + targetModel, + options.selector, + options.thinkingLevel, + ), + ); + } + this.#host.settings.getStorage()?.recordModelUsage(`${targetModel.provider}/${targetModel.id}`); + + // Re-apply thinking for the newly selected model. Prefer the model's + // configured defaultLevel; otherwise preserve the current level (or auto). + this.#reapplyThinkingLevel(targetModel.thinking?.defaultLevel); + await this.#host.syncAfterModelChange(previousEditMode); + return { switched: true }; + } + + /** + * Set model temporarily (for this session only). + * Validates that a credential source is configured (synchronously, without + * refreshing OAuth or running command-backed key programs), saves to session + * log but NOT to settings. + * @throws Error if no API key available for the model + */ + async setModelTemporary( + model: Model, + thinkingLevel?: ConfiguredThinkingLevel, + options?: { ephemeral?: boolean }, + ): Promise { + const previousEditMode = this.#host.resolveActiveEditMode(); + if (!this.#host.modelRegistry.hasConfiguredAuth(model)) { + throw new Error(`No API key for ${model.provider}/${model.id}`); + } + + const targetModel = await this.#host.modelRegistry.refreshSelectedModelMetadata(model); + + this.#host.modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); + this.#host.clearActiveRetryFallback(); + this.#host.setModelWithProviderSessionReset(targetModel); + this.#host.sessionManager.appendModelChange( + `${targetModel.provider}/${targetModel.id}`, + options?.ephemeral ? EPHEMERAL_MODEL_CHANGE_ROLE : "temporary", + ); + this.#host.settings.getStorage()?.recordModelUsage(`${targetModel.provider}/${targetModel.id}`); + + // Apply explicit thinking level if given; otherwise prefer the model's + // configured defaultLevel; otherwise re-clamp the current level (or auto). + if (thinkingLevel !== undefined) { + this.setThinkingLevel(thinkingLevel); + } else { + this.#reapplyThinkingLevel(targetModel.thinking?.defaultLevel); + } + await this.#host.syncAfterModelChange(previousEditMode); + } + + /** + * Cycle to next/previous model. + * Uses scoped models (from --models flag) if available, otherwise all available models. + * @param direction - "forward" (default) or "backward" + * @returns The new model info, or undefined if only one model available + */ + async cycleModel(direction: "forward" | "backward" = "forward"): Promise { + if (this.#scopedModels.length > 0) { + return this.#cycleScopedModel(direction); + } + return this.#cycleAvailableModel(direction); + } + + /** + * Resolve the configured role models in the given order plus the index of + * the currently active one. Roles that have no configured model, or whose + * configured model is not currently available, are skipped. The `default` + * role falls back to the active model when no explicit assignment exists. + * + * Returns `undefined` only when there is no current model or no available + * models at all; an empty `models` array is never returned (callers should + * still guard on `models.length`). + */ + getRoleModelCycle(roleOrder: readonly string[]): RoleModelCycle | undefined { + const availableModels = this.#host.modelRegistry.getAvailable(); + if (availableModels.length === 0) return undefined; + + const currentModel = this.#model; + if (!currentModel) return undefined; + const matchPreferences = getModelMatchPreferences(this.#host.settings); + const models: ResolvedRoleModel[] = []; + + for (const role of roleOrder) { + const roleModelStr = + role === "default" + ? (this.#host.settings.getModelRole("default") ?? `${currentModel.provider}/${currentModel.id}`) + : this.#host.settings.getModelRole(role); + if (!roleModelStr) continue; + + const resolved = resolveModelRoleValue(roleModelStr, availableModels, { + settings: this.#host.settings, + matchPreferences, + }); + if (!resolved.model) continue; + + models.push({ + role, + model: resolved.model, + thinkingLevel: resolved.thinkingLevel, + explicitThinkingLevel: resolved.explicitThinkingLevel, + }); + } + + if (models.length === 0) return undefined; + + // Trust the recorded role only while its resolved model still IS the + // active model. A model switch through another surface (alt+m, retry + // fallback, /model) or a role re-configuration leaves the recorded role + // pointing at a model the session no longer runs; cycling from that + // stale slot lands on the wrong neighbor and reads as a skipped entry. + const lastRole = this.#host.sessionManager.getLastModelChangeRole(); + let currentIndex = lastRole ? models.findIndex(entry => entry.role === lastRole) : -1; + if (currentIndex !== -1 && !modelsAreEqual(models[currentIndex].model, currentModel)) { + currentIndex = -1; + } + if (currentIndex === -1) { + currentIndex = models.findIndex(entry => modelsAreEqual(entry.model, currentModel)); + } + if (currentIndex === -1) currentIndex = 0; + + return { models, currentIndex }; + } + + /** + * Apply a resolved role model as the active model without changing global + * settings. Shared with role cycling and the plan-approval model slider. + */ + async applyRoleModel(entry: ResolvedRoleModel): Promise { + await this.setModel(entry.model, entry.role); + if (entry.explicitThinkingLevel && entry.thinkingLevel !== undefined) { + this.setThinkingLevel(entry.thinkingLevel); + } + } + + /** + * Cycle through configured role models in a fixed order. + * Skips missing roles and changes only the active session model. + * @param roleOrder - Order of roles to cycle through (e.g., ["slow", "default", "smol"]) + * @param direction - "forward" (default) or "backward" + */ + async cycleRoleModels( + roleOrder: readonly string[], + direction: "forward" | "backward" = "forward", + ): Promise { + const cycle = this.getRoleModelCycle(roleOrder); + if (!cycle || cycle.models.length <= 1) return undefined; + + const step = direction === "backward" ? -1 : 1; + const next = cycle.models[(cycle.currentIndex + step + cycle.models.length) % cycle.models.length]; + + await this.applyRoleModel(next); + + return { model: next.model, thinkingLevel: this.thinkingLevel, role: next.role }; + } + + async #getScopedModelsWithApiKey(): Promise> { + const apiKeysByProvider = new Map(); + const result: Array<{ model: Model; thinkingLevel?: ThinkingLevel }> = []; + + for (const scoped of this.#scopedModels) { + const provider = scoped.model.provider; + let apiKey: string | undefined; + if (apiKeysByProvider.has(provider)) { + apiKey = apiKeysByProvider.get(provider); + } else { + apiKey = await this.#host.modelRegistry.getApiKeyForProvider(provider, this.#host.sessionId()); + apiKeysByProvider.set(provider, apiKey); + } + + if (apiKey) { + result.push(scoped); + } + } + + return result; + } + + async #cycleScopedModel(direction: "forward" | "backward"): Promise { + const previousEditMode = this.#host.resolveActiveEditMode(); + const scopedModels = await this.#getScopedModelsWithApiKey(); + if (scopedModels.length <= 1) return undefined; + + const currentModel = this.#model; + let currentIndex = scopedModels.findIndex(sm => modelsAreEqual(sm.model, currentModel)); + + if (currentIndex === -1) currentIndex = 0; + const len = scopedModels.length; + const nextIndex = direction === "forward" ? (currentIndex + 1) % len : (currentIndex - 1 + len) % len; + const next = scopedModels[nextIndex]; + + // Apply model + this.#host.modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(next.model)); + this.#host.clearActiveRetryFallback(); + this.#host.setModelWithProviderSessionReset(next.model); + this.#host.sessionManager.appendModelChange(`${next.model.provider}/${next.model.id}`); + this.#host.settings.getStorage()?.recordModelUsage(`${next.model.provider}/${next.model.id}`); + + // Apply the scoped model's configured thinking level, preserving auto. + this.setThinkingLevel(this.#autoThinking ? AUTO_THINKING : next.thinkingLevel); + await this.#host.syncAfterModelChange(previousEditMode); + + return { model: next.model, thinkingLevel: this.thinkingLevel, isScoped: true }; + } + + async #cycleAvailableModel(direction: "forward" | "backward"): Promise { + const previousEditMode = this.#host.resolveActiveEditMode(); + const availableModels = this.#host.modelRegistry.getAvailable(); + if (availableModels.length <= 1) return undefined; + + const currentModel = this.#model; + let currentIndex = availableModels.findIndex(m => modelsAreEqual(m, currentModel)); + + if (currentIndex === -1) currentIndex = 0; + const len = availableModels.length; + const nextIndex = direction === "forward" ? (currentIndex + 1) % len : (currentIndex - 1 + len) % len; + const nextModel = availableModels[nextIndex]; + + const apiKey = await this.#host.modelRegistry.getApiKey(nextModel, this.#host.sessionId()); + if (!apiKey) { + throw new Error(`No API key for ${nextModel.provider}/${nextModel.id}`); + } + + this.#host.modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(nextModel)); + this.#host.clearActiveRetryFallback(); + this.#host.setModelWithProviderSessionReset(nextModel); + this.#host.sessionManager.appendModelChange(`${nextModel.provider}/${nextModel.id}`); + this.#host.settings.getStorage()?.recordModelUsage(`${nextModel.provider}/${nextModel.id}`); + // Re-apply the current thinking level (or auto) for the newly selected model + this.#reapplyThinkingLevel(); + await this.#host.syncAfterModelChange(previousEditMode); + + return { model: nextModel, thinkingLevel: this.thinkingLevel, isScoped: false }; + } + + /** + * Get all available models with valid API keys, filtered by `enabledModels` when configured. + * See {@link filterAvailableModelsByEnabledPatterns} for supported pattern forms and limitations. + */ + getAvailableModels(): Model[] { + const all = this.#host.modelRegistry.getAvailable(); + const patterns = this.#host.settings.get("enabledModels"); + if (!patterns || patterns.length === 0) return all; + return filterAvailableModelsByEnabledPatterns(all, patterns, this.#host.settings); + } + + // ========================================================================= + // Thinking Level Management + // ========================================================================= + + #applyThinkingLevelToAgent(level: ThinkingLevel | undefined): void { + this.#host.agent.setThinkingLevel(toReasoningEffort(level)); + this.#host.agent.setDisableReasoning(shouldDisableReasoning(level)); + } + + /** + * Set the thinking level. `auto` enables per-turn classification. Entering + * auto writes its provisional level plus `configured: "auto"` immediately, + * giving external readers an authoritative selection receipt before the next + * user turn. Later classifications persist only changed concrete resolutions. + */ + setThinkingLevel(level: ConfiguredThinkingLevel | undefined, persist: boolean = false): void { + if (level === AUTO_THINKING) { + const provisional = resolveProvisionalAutoLevel(this.#model); + const wasAuto = this.#autoThinking; + const previousLevel = this.#thinkingLevel; + this.#autoThinking = true; + this.#autoResolvedLevel = undefined; + this.#thinkingLevel = provisional; + if (!wasAuto) { + this.#host.clearInheritedProviderPromptCacheKey(); + } + this.#applyThinkingLevelToAgent(provisional); + if (persist) { + this.#host.settings.set("defaultThinkingLevel", AUTO_THINKING); + } + const isChanging = !wasAuto || previousLevel !== provisional; + if (isChanging) { + this.#host.sessionManager.appendThinkingLevelChange(provisional, AUTO_THINKING); + this.#host.emit({ type: "thinking_level_changed", thinkingLevel: provisional, configured: AUTO_THINKING }); + } + return; + } + + const wasAuto = this.#autoThinking; + this.#autoThinking = false; + this.#autoResolvedLevel = undefined; + const effectiveLevel = resolveThinkingLevelForModel(this.#model, level); + // Leaving auto must persist even when the resolved effort is unchanged (e.g. + // auto resolved to medium, then the user pins medium): otherwise the latest + // session entry keeps `configured: "auto"` and resume re-enables auto. + const isChanging = wasAuto || effectiveLevel !== this.#thinkingLevel; + + this.#thinkingLevel = effectiveLevel; + this.#applyThinkingLevelToAgent(effectiveLevel); + + if (isChanging) { + this.#host.clearInheritedProviderPromptCacheKey(); + this.#host.sessionManager.appendThinkingLevelChange(effectiveLevel, effectiveLevel); + if (persist && effectiveLevel !== undefined && effectiveLevel !== ThinkingLevel.Off) { + this.#host.settings.set("defaultThinkingLevel", effectiveLevel); + } + this.#host.emit({ type: "thinking_level_changed", thinkingLevel: effectiveLevel }); + } + } + + /** + * Re-apply the active thinking selection after a model change. Preserves `auto` + * (re-clamping the provisional level to the new model); otherwise re-applies the + * preferred default or the current effective level. + */ + #reapplyThinkingLevel(preferredDefault?: ThinkingLevel): void { + this.setThinkingLevel(this.#autoThinking ? AUTO_THINKING : (preferredDefault ?? this.#thinkingLevel)); + } + + /** + * Cycle to next thinking level: off → auto → minimal..max → off. + * @returns New selector, or undefined if model doesn't support thinking + */ + cycleThinkingLevel(): ConfiguredThinkingLevel | undefined { + if (!this.#model?.reasoning) return undefined; + + const levels: ConfiguredThinkingLevel[] = [ + ThinkingLevel.Off, + AUTO_THINKING, + ...this.getAvailableThinkingLevels(), + ]; + const configured = this.configuredThinkingLevel(); + const currentLevel = configured === ThinkingLevel.Inherit ? ThinkingLevel.Off : configured; + const currentIndex = currentLevel ? levels.indexOf(currentLevel) : -1; + const nextIndex = (currentIndex + 1) % levels.length; + const nextLevel = levels[nextIndex]; + if (!nextLevel) return undefined; + + this.setThinkingLevel(nextLevel); + return nextLevel; + } + + /** Timeout (ms) for per-turn auto-thinking classification before falling back. */ + static readonly #AUTO_THINKING_TIMEOUT_MS = 4000; + + /** + * Classify the current user turn and set the effective thinking level for it. + * Bounded by a timeout + abort; on any failure (no smol model, timeout, parse + * error) it falls back to the provisional concrete level and continues. Never + * throws into the turn, and never clears `#autoThinking` (auto stays active). + */ + async applyAutoThinkingLevel(promptText: string, generation: number): Promise { + const model = this.#model; + if (!model?.reasoning) return; + // Models with reasoning but no controllable effort surface (devin-agent + // Cascade routes effort via sibling model ids, not a wire param) have + // nothing to pick — skip classification rather than discard its result. + if (getSupportedEfforts(model).length === 0) return; + + let resolved: Effort | undefined; + if (this.#host.magicKeywordEnabled("ultrathink") && containsUltrathink(promptText)) { + // The user explicitly asked for maximum thinking; bypass the classifier + // (and its xhigh auto ceiling) and jump straight to the highest + // supported level for this model. + resolved = clampAutoThinkingEffort(model, Effort.Max); + } else { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), ModelControls.#AUTO_THINKING_TIMEOUT_MS); + try { + resolved = await classifyDifficulty(promptText, { + settings: this.#host.settings, + registry: this.#host.modelRegistry, + model, + sessionId: this.#host.sessionId(), + signal: controller.signal, + metadataResolver: provider => this.#host.agent.metadataForProvider(provider), + }); + } catch (error) { + logger.debug("auto-thinking: classification failed; using fallback level", { + error: error instanceof Error ? error.message : String(error), + }); + } finally { + clearTimeout(timer); + } + } + + // Drop the result if the turn was aborted/superseded while classifying. + if (this.#host.promptGeneration() !== generation || !this.#autoThinking) return; + + const effort = resolved ?? resolveProvisionalAutoLevel(model); + if (effort === undefined) return; + const shouldPersistResolution = this.#thinkingLevel !== effort; + this.#autoResolvedLevel = effort; + this.#thinkingLevel = effort; + this.#applyThinkingLevelToAgent(effort); + if (shouldPersistResolution) { + this.#host.sessionManager.appendThinkingLevelChange(effort, AUTO_THINKING); + } + this.#host.emit({ + type: "thinking_level_changed", + thinkingLevel: effort, + configured: AUTO_THINKING, + resolved: effort, + }); + } + + /** + * True when the currently selected model's family is set to `priority` — the + * `/fast` on/off state for the active model. Returns false when no model is + * selected or the model exposes no service-tier family (e.g. Fireworks, which + * has its own Providers › Fireworks Tier toggle). + * + * For "is priority actually applied to the next request?" use + * {@link isFastModeActive} instead. + */ + isFastModeEnabled(): boolean { + const family = this.#model ? serviceTierFamily(this.#model) : undefined; + return family ? this.#serviceTierByFamily[family] === "priority" : false; + } + + /** + * True when `priority` is actually realized on the wire for the currently + * selected model (OpenAI/Google `service_tier`, direct Anthropic fast mode, + * or Fireworks priority). Returns false for tiers the active model can't + * realize and when no model is selected. + */ + isFastModeActive(): boolean { + const model = this.#model; + return !!model && realizesPriorityServiceTier(this.effectiveServiceTier(model), model); + } + + /** + * Effective wire service-tier for a request to `model`. Fireworks models take + * the Priority serving path only when the Providers › Fireworks Tier setting + * is `"priority"` (and never for `-fast` variants, whose Fast serving path is + * mutually exclusive with Priority). Every other model resolves the live + * per-family tier map down to the entry for its family. + */ + effectiveServiceTier(model: Model | undefined = this.#model): ServiceTier | undefined { + if (model?.provider === "fireworks") { + return this.#host.settings.get("providers.fireworksTier") === "priority" && !isFireworksFastModelId(model.id) + ? "priority" + : undefined; + } + if (!model) return undefined; + return resolveModelServiceTier(this.#serviceTierByFamily, model); + } + + /** The live per-family tier map, or `null` when empty (for session persistence). */ + serviceTierEntry(): ServiceTierByFamily | null { + return Object.keys(this.#serviceTierByFamily).length > 0 ? this.#serviceTierByFamily : null; + } + + /** Set one family's tier (or clear it with `undefined`); persists the change. */ + setServiceTierFamily(family: ServiceTierFamily, tier: ServiceTier | undefined): void { + if (this.#serviceTierByFamily[family] === tier) return; + const next: ServiceTierByFamily = { ...this.#serviceTierByFamily }; + if (tier) next[family] = tier; + else delete next[family]; + this.#applyServiceTierByFamily(next); + } + + /** Replace the whole per-family tier map; persists + re-arms Anthropic fast mode. */ + #applyServiceTierByFamily(next: ServiceTierByFamily): void { + // Re-arming Anthropic priority clears the per-session fast-mode auto-disable + // so the next request actually carries `speed: "fast"` again. + if (next.anthropic === "priority" && this.#serviceTierByFamily.anthropic !== "priority") { + clearAnthropicFastModeFallback(this.#host.providerSessionState); + } + this.#serviceTierByFamily = next; + this.#host.sessionManager.appendServiceTierChange(this.serviceTierEntry()); + } + + /** + * `/fast on|off` targets the family of the currently selected model: it sets + * (or clears) that family's `priority` tier. Returns `false` when the model + * has no service-tier family, so callers can report that fast mode is + * unavailable instead of claiming success. + */ + setFastMode(enabled: boolean): boolean { + const family = this.#model ? serviceTierFamily(this.#model) : undefined; + if (!family) { + this.#host.emitNotice( + "info", + "The current model has no service-tier control for /fast to toggle.", + "priority", + ); + return false; + } + if (!enabled) { + if (this.#serviceTierByFamily[family] === "priority") this.setServiceTierFamily(family, undefined); + return true; + } + this.setServiceTierFamily(family, "priority"); + return true; + } + + toggleFastMode(): boolean { + if (!this.setFastMode(!this.isFastModeEnabled())) return false; + return this.isFastModeEnabled(); + } + + /** + * Get available thinking levels for current model. + */ + getAvailableThinkingLevels(): ReadonlyArray { + if (!this.#model) return []; + return getSupportedEfforts(this.#model); + } +} diff --git a/packages/coding-agent/src/session/prewalk.ts b/packages/coding-agent/src/session/prewalk.ts new file mode 100644 index 000000000..9dcfa1a4b --- /dev/null +++ b/packages/coding-agent/src/session/prewalk.ts @@ -0,0 +1,252 @@ +import type { Agent, AgentMessage, AgentToolResult, AgentTurnEndContext } from "@oh-my-pi/pi-agent-core"; +import { invalidateMessageCache } from "@oh-my-pi/pi-agent-core/compaction"; +import type { Model } from "@oh-my-pi/pi-ai"; +import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; +import { prompt } from "@oh-my-pi/pi-utils"; +import type { LocalProtocolOptions } from "../internal-urls"; +import { resolveApprovedPlan } from "../plan-mode/approved-plan"; +import { listPlanFiles, readPlanFile } from "../plan-mode/plan-files"; +import type { PlanModeState } from "../plan-mode/state"; +import planYoloHandoffPrompt from "../prompts/system/plan-yolo-handoff.md" with { type: "text" }; +import prewalkChecklistPrompt from "../prompts/system/prewalk-checklist.md" with { type: "text" }; +import prewalkContinuePrompt from "../prompts/system/prewalk-continue.md" with { type: "text" }; +import prewalkPlanPrompt from "../prompts/system/prewalk-plan.md" with { type: "text" }; +import type { ConfiguredThinkingLevel } from "../thinking"; +import type { PlanProposalHandler } from "../tools/resolve"; +import { ToolError } from "../tools/tool-errors"; +import type { PlanYolo, Prewalk } from "./agent-session-types"; +import type { SessionManager } from "./session-manager"; + +const PREWALK_PLAN_MESSAGE_TYPE = "prewalk-plan"; +const PREWALK_CONTINUE_MESSAGE_TYPE = "prewalk-continue"; +const PREWALK_CHECKLIST_MESSAGE_TYPE = "prewalk-checklist"; +const PREWALK_ACTION_TOOLS: Record = { + edit: true, + write: true, +}; +const PLAN_YOLO_HANDOFF_MESSAGE_TYPE = "plan-yolo-handoff"; + +/** Capabilities the prewalk coordinator borrows from its owning session. */ +export interface PrewalkCoordinatorHost { + agent: Agent; + sessionManager: SessionManager; + model(): Model | undefined; + emitNotice(level: "info" | "warning" | "error", message: string, source?: string): void; + setModelTemporary( + model: Model, + thinkingLevel?: ConfiguredThinkingLevel, + options?: { ephemeral?: boolean }, + ): Promise; + setActiveToolsByName(names: string[]): Promise; + getActiveToolNames(): string[]; + getEnabledToolNames(): string[]; + hasBuiltInTool(name: string): boolean; + getPlanModeState(): PlanModeState | undefined; + setPlanModeState(state: PlanModeState | undefined): void; + getPlanReferencePath(): string; + setPlanProposalHandler(handler: PlanProposalHandler | null): void; + waitForSessionMessagePersistence(message: AgentMessage): Promise; + localProtocolOptions(): LocalProtocolOptions; +} + +/** Initial state for prewalk and plan-yolo startup flows. */ +export interface PrewalkCoordinatorOptions { + prewalk?: Prewalk; + planYolo?: PlanYolo; +} + +/** Coordinates one-way model prewalks and automatic plan-yolo handoffs. */ +export class PrewalkCoordinator { + readonly #host: PrewalkCoordinatorHost; + #prewalk: Prewalk | undefined; + #planInjected = false; + #continuePending = false; + #todoSeen = false; + #planYolo: PlanYolo | undefined; + #planYoloPreviousTools: string[] | undefined; + #planYoloArmed = false; + + constructor(host: PrewalkCoordinatorHost, options: PrewalkCoordinatorOptions = {}) { + this.#host = host; + this.#prewalk = options.prewalk; + this.#planYolo = options.planYolo; + } + + /** Current prewalk target, if the one-way switch remains armed. */ + get state(): Prewalk | undefined { + return this.#prewalk; + } + + /** Advances the one-way prewalk switch at a completed assistant-turn boundary. */ + async advanceAtTurnEnd(liveMessages: AgentMessage[], context: AgentTurnEndContext | undefined): Promise { + const prewalk = this.#prewalk; + if (!prewalk || context?.message.role !== "assistant") return; + if (context.toolResults.some(result => result.toolName === "todo" && !result.isError)) this.#todoSeen = true; + + const hasToolResults = context.toolResults.length > 0; + if (this.#planInjected && hasToolResults) { + this.#continuePending = true; + } else if (this.#continuePending) { + this.#continuePending = false; + this.#host.agent.steer({ + role: "custom", + customType: PREWALK_CONTINUE_MESSAGE_TYPE, + content: prewalkContinuePrompt, + attribution: "agent", + display: false, + timestamp: Date.now(), + }); + } + + const todoGateOpen = this.#todoSeen || !this.#host.getActiveToolNames().includes("todo"); + const action = todoGateOpen + ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) + : undefined; + if (!action) { + if (!this.#planInjected) { + this.#planInjected = true; + this.#continuePending = true; + this.#host.agent.steer({ + role: "custom", + customType: PREWALK_PLAN_MESSAGE_TYPE, + content: prewalkPlanPrompt, + display: false, + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.emitNotice("info", "Prewalk: injected deep-plan nudge.", "prewalk"); + } + return; + } + + await this.#host.waitForSessionMessagePersistence(context.message); + for (const toolResult of context.toolResults) { + await this.#host.waitForSessionMessagePersistence(toolResult); + } + this.#scrubPlanNudge(liveMessages); + const target = prewalk.target; + const currentModel = this.#host.model(); + if (currentModel && modelsAreEqual(currentModel, target)) { + this.#prewalk = undefined; + return; + } + await this.#host.setModelTemporary(target, prewalk.thinkingLevel, { ephemeral: true }); + this.#prewalk = undefined; + this.#host.emitNotice( + "info", + `Prewalk: switched to ${target.provider}/${target.id} after first ${action.toolName} call.`, + "prewalk", + ); + this.#host.agent.steer({ + role: "custom", + customType: PREWALK_CHECKLIST_MESSAGE_TYPE, + content: prewalkChecklistPrompt, + attribution: "agent", + display: false, + timestamp: Date.now(), + }); + } + + /** Arms a prewalk immediately for an explicit slash-command request. */ + arm(target: Model, thinkingLevel?: ConfiguredThinkingLevel): void { + if (this.#prewalk) { + this.#host.emitNotice( + "info", + `Prewalk: already armed for ${this.#prewalk.target.provider}/${this.#prewalk.target.id}, waiting for the first edit/write.`, + "prewalk", + ); + return; + } + this.#prewalk = { target, thinkingLevel }; + this.#planInjected = true; + this.#continuePending = true; + this.#host.agent.steer({ + role: "custom", + customType: PREWALK_PLAN_MESSAGE_TYPE, + content: prewalkPlanPrompt, + display: false, + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.emitNotice( + "info", + `Prewalk: armed for ${target.provider}/${target.id} — will switch at the first edit/write once the todo list exists.`, + "prewalk", + ); + } + + /** Lazily enables plan-yolo's plan phase before the first prompt is built. */ + async armPlanYoloIfNeeded(): Promise { + if (!this.#planYolo || this.#planYoloArmed) return; + this.#planYoloArmed = true; + const previousTools = this.#host.getEnabledToolNames(); + const augmentations = this.#host.hasBuiltInTool("write") ? ["write"] : []; + await this.#host.setActiveToolsByName([...new Set([...previousTools, ...augmentations])]); + this.#planYoloPreviousTools = previousTools; + this.#host.setPlanModeState({ + enabled: true, + planFilePath: this.#host.getPlanReferencePath() || "local://PLAN.md", + workflow: "parallel", + }); + this.#host.setPlanProposalHandler(title => this.#finalizePlanYoloProposal(title)); + } + + #scrubPlanNudge(liveMessages: AgentMessage[]): void { + if (!this.#planInjected) return; + const isPlanNudge = (message: AgentMessage): boolean => + message.role === "custom" && message.customType === PREWALK_PLAN_MESSAGE_TYPE; + for (let index = liveMessages.length - 1; index >= 0; index--) { + if (!isPlanNudge(liveMessages[index])) continue; + invalidateMessageCache(liveMessages[index]); + liveMessages.splice(index, 1); + } + const stateMessages = this.#host.agent.state.messages; + const filtered = stateMessages.filter(message => !isPlanNudge(message)); + if (filtered.length !== stateMessages.length) this.#host.agent.replaceMessages(filtered); + } + + async #finalizePlanYoloProposal(title: string): Promise> { + const planYolo = this.#planYolo; + const state = this.#host.getPlanModeState(); + if (!planYolo || !state?.enabled) throw new ToolError("Plan mode is not active."); + const { planFilePath, title: resolvedTitle } = await resolveApprovedPlan({ + suppliedTitle: title, + statePlanFilePath: state.planFilePath, + readPlan: url => + readPlanFile(url, { + localProtocolOptions: this.#host.localProtocolOptions(), + cwd: this.#host.sessionManager.getCwd(), + }), + listPlanFiles: () => listPlanFiles({ localProtocolOptions: this.#host.localProtocolOptions() }), + }); + this.#host.setPlanModeState(undefined); + const previousTools = this.#planYoloPreviousTools; + try { + if (previousTools) await this.#host.setActiveToolsByName(previousTools); + } catch (error) { + this.#host.setPlanModeState(state); + throw error; + } + this.#host.setPlanProposalHandler(null); + this.#planYolo = undefined; + this.#planYoloPreviousTools = undefined; + await this.#host.setModelTemporary(planYolo.target, planYolo.thinkingLevel, { ephemeral: true }); + this.#host.emitNotice( + "info", + `Plan-yolo: plan approved, switched to ${planYolo.target.provider}/${planYolo.target.id} to implement "${resolvedTitle}".`, + "plan-yolo", + ); + this.#host.agent.steer({ + role: "custom", + customType: PLAN_YOLO_HANDOFF_MESSAGE_TYPE, + content: prompt.render(planYoloHandoffPrompt, { planFilePath, title: resolvedTitle }), + attribution: "agent", + display: false, + timestamp: Date.now(), + }); + return { + content: [{ type: "text", text: `Plan approved. Implementing now with ${planYolo.target.id}.` }], + details: { planFilePath, title: resolvedTitle, planExists: true }, + }; + } +} diff --git a/packages/coding-agent/src/session/queued-messages.ts b/packages/coding-agent/src/session/queued-messages.ts new file mode 100644 index 000000000..49434a4ce --- /dev/null +++ b/packages/coding-agent/src/session/queued-messages.ts @@ -0,0 +1,93 @@ +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage, ImageContent } from "@oh-my-pi/pi-ai"; +import type { RestoredQueuedMessage } from "./agent-session-types"; +import { type CustomMessage, readQueueChipText } from "./messages"; + +function queuedTextContent(message: AgentMessage): string | undefined { + if (!("content" in message)) return undefined; + const content = message.content; + if (typeof content === "string") return content; + for (const part of content) { + if (part.type === "text") return part.text; + } + return undefined; +} + +function queuedImageContent(message: AgentMessage): ImageContent[] | undefined { + if (!("content" in message) || typeof message.content === "string") return undefined; + const images: ImageContent[] = []; + for (const part of message.content) { + if (part.type === "image" && typeof part.data === "string" && typeof part.mimeType === "string") { + images.push(part); + } + } + return images.length > 0 ? images : undefined; +} + +/** Whether a queued message should render in the queue UI. */ +export function isDisplayableQueuedMessage(message: AgentMessage): boolean { + return !(message.role === "custom" && message.display === false); +} + +/** Whether a queued message is an advisor card. */ +export function isAdvisorCard(message: AgentMessage): message is CustomMessage { + return message.role === "custom" && message.customType === "advisor"; +} + +/** Whether a message is a terminal assistant answer containing text and no tools. */ +export function isTerminalTextAssistantAnswer(message: AgentMessage | undefined): message is AssistantMessage { + if (message?.role !== "assistant" || message.stopReason !== "stop") return false; + let hasText = false; + for (const part of message.content) { + if (part.type === "toolCall") return false; + if (part.type === "text") { + if (part.text.trim().length > 0) hasText = true; + continue; + } + if (part.type === "thinking" || part.type === "redactedThinking" || part.type === "fallback") continue; + return false; + } + return hasText; +} + +/** Whether queued content was authored by the user and can be restored to the editor. */ +export function isUserQueuedMessage(message: AgentMessage): boolean { + if (message.role === "user") return true; + return message.role === "custom" && message.attribution === "user" && message.display !== false; +} + +/** Hidden magic-keyword notices queued alongside a user prompt. */ +export const MAGIC_KEYWORD_NOTICE_TYPES: Record = { + "ultrathink-notice": true, + "orchestrate-notice": true, + "workflow-notice": true, +}; + +/** Hidden companion carrying vision descriptions for a text-only model. */ +export const IMAGE_ATTACHMENT_DESCRIPTION_TYPE = "image-attachment-description"; + +/** Whether a hidden queued message is a companion of an adjacent user prompt. */ +export function isHiddenUserCompanion(message: AgentMessage): boolean { + return ( + message.role === "custom" && + message.attribution === "user" && + message.display === false && + (MAGIC_KEYWORD_NOTICE_TYPES[message.customType] === true || + message.customType === IMAGE_ATTACHMENT_DESCRIPTION_TYPE) + ); +} + +/** Human-readable text shown for a queued-message chip. */ +export function queueChipText(message: AgentMessage): string { + if (message.role === "custom") { + return readQueueChipText(message.details) ?? queuedTextContent(message) ?? ""; + } + const text = queuedTextContent(message) ?? ""; + if (text) return text; + return queuedImageContent(message) ? "[Image]" : ""; +} + +/** Converts a queued user message to editor-restorable content. */ +export function toRestoredQueuedMessage(message: AgentMessage): RestoredQueuedMessage { + return { text: queueChipText(message), images: queuedImageContent(message) }; +} diff --git a/packages/coding-agent/src/session/retry-fallback-chains.ts b/packages/coding-agent/src/session/retry-fallback-chains.ts new file mode 100644 index 000000000..61ff1fca8 --- /dev/null +++ b/packages/coding-agent/src/session/retry-fallback-chains.ts @@ -0,0 +1,257 @@ +import type { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import type { Model } from "@oh-my-pi/pi-ai"; +import { logger } from "@oh-my-pi/pi-utils"; +import type { ModelRegistry } from "../config/model-registry"; +import { formatModelSelectorValue, formatModelStringWithRouting, parseModelString } from "../config/model-resolver"; +import type { Settings } from "../config/settings"; +import { type ConfiguredThinkingLevel, concreteThinkingLevel } from "../thinking"; + +/** Configured fallback chains keyed by role or model selector. */ +export type RetryFallbackChains = Record; + +/** Policy controlling restoration of a fallback chain's primary model. */ +export type RetryFallbackRevertPolicy = "never" | "cooldown-expiry"; + +/** Parsed model selector used by retry fallback resolution. */ +export interface RetryFallbackSelector { + raw: string; + provider: string; + id: string; + thinkingLevel: ThinkingLevel | undefined; +} + +/** Active retry fallback state retained until the primary can be restored. */ +export interface ActiveRetryFallbackState { + /** Chain key that produced this fallback: a model-role name or a model-selector key. */ + role: string; + originalSelector: string; + originalThinkingLevel: ConfiguredThinkingLevel | undefined; + lastAppliedFallbackThinkingLevel: ConfiguredThinkingLevel | undefined; + pinned: boolean; +} + +const RETRY_BACKOFF_MAX_DELAY_MS = 8_000; +const RETRY_BACKOFF_JITTER_RATIO = 0.25; + +/** Calculates capped exponential retry delay with downward jitter. */ +export function calculateRetryBackoffDelayMs(baseDelayMs: number, attempt: number): number { + const cappedDelayMs = Math.min(Math.max(0, baseDelayMs) * 2 ** Math.max(0, attempt - 1), RETRY_BACKOFF_MAX_DELAY_MS); + const jitter = 1 - Math.random() * RETRY_BACKOFF_JITTER_RATIO; + return cappedDelayMs * jitter; +} + +/** Parses a configured retry fallback selector. */ +export function parseRetryFallbackSelector( + selector: string, + modelLookup?: { find(provider: string, id: string): Model | undefined }, +): RetryFallbackSelector | undefined { + const trimmed = selector.trim(); + if (!trimmed) return undefined; + const parsed = parseModelString(trimmed, { + allowMaxSuffix: true, + allowAutoAlias: true, + isLiteralModelId: (provider, id) => modelLookup?.find(provider, id) !== undefined, + }); + if (!parsed) return undefined; + return { + raw: trimmed, + provider: parsed.provider, + id: parsed.id, + thinkingLevel: concreteThinkingLevel(parsed.thinkingLevel), + }; +} + +/** Whether a fallback-chain key is a model selector rather than a role. */ +export function isRetryFallbackModelKey(key: string): boolean { + return key.includes("/"); +} + +/** Whether a fallback-chain key or entry is a provider wildcard. */ +export function isRetryFallbackWildcardKey(key: string): boolean { + return key.endsWith("/*"); +} + +/** Splits a wildcard selector into provider and optional model-id prefix. */ +export function parseRetryFallbackWildcard( + key: string, + isKnownProvider: (provider: string) => boolean, +): { provider: string; idPrefix: string | undefined } { + const template = key.slice(0, -2); + const slash = template.indexOf("/"); + if (slash < 0 || isKnownProvider(template)) return { provider: template, idPrefix: undefined }; + return { provider: template.slice(0, slash), idPrefix: template.slice(slash + 1) }; +} + +/** Formats a concrete model and thinking level as a fallback selector. */ +export function formatRetryFallbackSelector(model: Model, thinkingLevel: ThinkingLevel | undefined): string { + return formatModelSelectorValue(formatModelStringWithRouting(model), thinkingLevel); +} + +/** Formats the model-only portion of a parsed fallback selector. */ +export function formatRetryFallbackBaseSelector(selector: RetryFallbackSelector): string { + return `${selector.provider}/${selector.id}`; +} + +/** Whether a provider is registered or configured for discovery. */ +export function isKnownProvider(modelRegistry: ModelRegistry, provider: string): boolean { + return modelRegistry.hasProvider(provider); +} + +/** Resolves configured fallback chains, applying the default chain to named roles. */ +export function getRetryFallbackChains(settings: Settings): RetryFallbackChains { + const configuredChains = settings.get("retry.fallbackChains"); + if (!configuredChains || typeof configuredChains !== "object") return {}; + const chains: RetryFallbackChains = { ...configuredChains }; + const defaultChain = chains.default; + if (Array.isArray(defaultChain)) { + for (const role in settings.getModelRoles()) { + if (role !== "default" && chains[role] === undefined) chains[role] = defaultChain; + } + } + return chains; +} + +/** Validates configured fallback chains and reports each warning. */ +export function validateRetryFallbackChains( + settings: Settings, + modelRegistry: ModelRegistry, + warn: (message: string) => void, +): void { + const configuredChains = settings.get("retry.fallbackChains"); + if (configuredChains === undefined) return; + const report = (message: string) => { + logger.warn(message); + warn(message); + }; + if (!configuredChains || typeof configuredChains !== "object" || Array.isArray(configuredChains)) { + report("retry.fallbackChains must be a mapping of role names or model selectors to selector arrays."); + return; + } + + for (const key in configuredChains) { + const chain = configuredChains[key]; + const keyKind = isRetryFallbackModelKey(key) ? "model" : "role"; + if (keyKind === "model") { + if (isRetryFallbackWildcardKey(key)) { + const { provider } = parseRetryFallbackWildcard(key, candidate => + isKnownProvider(modelRegistry, candidate), + ); + if (!isKnownProvider(modelRegistry, provider)) { + report(`retry.fallbackChains wildcard key references unknown provider: ${key}`); + } + } else { + const parsedKey = parseRetryFallbackSelector(key, modelRegistry); + if (!parsedKey) { + report(`Invalid model selector key in retry.fallbackChains: ${key}`); + } else if (!modelRegistry.find(parsedKey.provider, parsedKey.id)) { + report(`retry.fallbackChains key references unknown model: ${key}`); + } + } + } + if (!Array.isArray(chain)) { + report(`Fallback chain for ${keyKind} '${key}' must be an array of selector strings.`); + continue; + } + for (const selectorStr of chain) { + if (typeof selectorStr !== "string") { + report(`Fallback chain for ${keyKind} '${key}' contains a non-string selector.`); + continue; + } + if (isRetryFallbackWildcardKey(selectorStr)) { + const { provider } = parseRetryFallbackWildcard(selectorStr, candidate => + isKnownProvider(modelRegistry, candidate), + ); + if (!isKnownProvider(modelRegistry, provider)) { + report(`Fallback chain for ${keyKind} '${key}' references unknown provider: ${selectorStr}`); + } + continue; + } + const parsed = parseRetryFallbackSelector(selectorStr, modelRegistry); + if (!parsed) { + report(`Invalid fallback selector format in ${keyKind} '${key}': ${selectorStr}`); + continue; + } + if (!modelRegistry.find(parsed.provider, parsed.id)) { + report(`Fallback chain for ${keyKind} '${key}' references unknown model: ${selectorStr}`); + } + } + } +} + +/** Returns the configured fallback-primary restoration policy. */ +export function getRetryFallbackRevertPolicy(settings: Settings): RetryFallbackRevertPolicy { + return settings.get("retry.fallbackRevertPolicy") === "never" ? "never" : "cooldown-expiry"; +} + +/** Resolves the primary selector represented by a fallback-chain key. */ +export function getRetryFallbackPrimarySelector( + settings: Settings, + modelRegistry: ModelRegistry, + role: string, +): RetryFallbackSelector | undefined { + if (isRetryFallbackWildcardKey(role)) return undefined; + if (isRetryFallbackModelKey(role)) return parseRetryFallbackSelector(role, modelRegistry); + const configuredSelector = settings.getModelRole(role); + return configuredSelector ? parseRetryFallbackSelector(configuredSelector, modelRegistry) : undefined; +} + +/** Parses one configured fallback-chain entry relative to the current model. */ +export function parseRetryFallbackChainEntry( + entry: string, + current: RetryFallbackSelector | undefined, + modelRegistry: ModelRegistry, +): RetryFallbackSelector | undefined { + if (isRetryFallbackWildcardKey(entry)) { + if (!current) return undefined; + const { provider, idPrefix } = parseRetryFallbackWildcard(entry, candidate => + isKnownProvider(modelRegistry, candidate), + ); + const bareId = current.id.slice(current.id.lastIndexOf("/") + 1); + let id: string; + if (idPrefix !== undefined) { + id = `${idPrefix}/${bareId}`; + } else if ( + bareId !== current.id && + !modelRegistry.find(provider, current.id) && + modelRegistry.find(provider, bareId) + ) { + // Aggregator → direct: the failing id carries a vendor prefix the + // target provider does not use (openrouter/google/x → google-vertex/x). + id = bareId; + } else { + id = current.id; + } + return { raw: `${provider}/${id}`, provider, id, thinkingLevel: undefined }; + } + return parseRetryFallbackSelector(entry, modelRegistry); +} + +/** Builds a fallback chain beginning with its effective primary selector. */ +export function getRetryFallbackEffectiveChain( + settings: Settings, + modelRegistry: ModelRegistry, + role: string, + currentSelector?: string, +): RetryFallbackSelector[] { + const parsedCurrent = currentSelector ? parseRetryFallbackSelector(currentSelector, modelRegistry) : undefined; + const seen = new Set(); + const chain: RetryFallbackSelector[] = []; + if (isRetryFallbackWildcardKey(role)) { + if (parsedCurrent) { + chain.push(parsedCurrent); + seen.add(parsedCurrent.raw); + } + } else { + const primarySelector = getRetryFallbackPrimarySelector(settings, modelRegistry, role); + if (!primarySelector) return []; + chain.push(primarySelector); + seen.add(primarySelector.raw); + } + for (const selector of getRetryFallbackChains(settings)[role] ?? []) { + const parsed = parseRetryFallbackChainEntry(selector, parsedCurrent, modelRegistry); + if (!parsed || seen.has(parsed.raw)) continue; + seen.add(parsed.raw); + chain.push(parsed); + } + return chain; +} diff --git a/packages/coding-agent/src/session/role-models.ts b/packages/coding-agent/src/session/role-models.ts new file mode 100644 index 000000000..d57b72c0c --- /dev/null +++ b/packages/coding-agent/src/session/role-models.ts @@ -0,0 +1,85 @@ +import type { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import type { Model } from "@oh-my-pi/pi-ai"; +import type { ModelRegistry } from "../config/model-registry"; +import { + extractExplicitThinkingSelector, + formatModelSelectorValue, + getModelMatchPreferences, + parseModelString, + type ResolvedModelRoleValue, + resolveModelRoleValue, +} from "../config/model-resolver"; +import type { Settings } from "../config/settings"; + +/** Formats a role assignment while preserving its explicit thinking selector. */ +export function formatRoleModelValue( + settings: Settings, + modelRegistry: ModelRegistry, + role: string, + model: Model, + selectorOverride?: string, + thinkingLevelOverride?: ThinkingLevel, +): string { + const modelKey = selectorOverride ?? `${model.provider}/${model.id}`; + if (thinkingLevelOverride !== undefined) return formatModelSelectorValue(modelKey, thinkingLevelOverride); + const existingRoleValue = settings.getModelRole(role); + if (!existingRoleValue) return modelKey; + const thinkingLevel = extractExplicitThinkingSelector(existingRoleValue, settings, { + isLiteralModelId: (provider, id) => modelRegistry.find(provider, id) !== undefined, + }); + return formatModelSelectorValue(modelKey, thinkingLevel); +} + +/** Resolves a configured model target relative to the current provider. */ +export function resolveConfiguredModelTarget( + configuredTarget: string | undefined, + currentModel: Model, + availableModels: Model[], +): Model | undefined { + const trimmedTarget = configuredTarget?.trim(); + if (!trimmedTarget) return undefined; + const parsed = parseModelString(trimmedTarget, { + allowMaxSuffix: true, + allowAutoAlias: true, + isLiteralModelId: (provider, id) => availableModels.some(model => model.provider === provider && model.id === id), + }); + if (parsed) { + const explicitModel = availableModels.find(model => model.provider === parsed.provider && model.id === parsed.id); + if (explicitModel) return explicitModel; + } + return availableModels.find(model => model.provider === currentModel.provider && model.id === trimmedTarget); +} + +/** Resolves a model's configured context-promotion target. */ +export function resolveContextPromotionConfiguredTarget( + currentModel: Model, + availableModels: Model[], +): Model | undefined { + return resolveConfiguredModelTarget(currentModel.contextPromotionTarget, currentModel, availableModels); +} + +/** Resolves a model's configured compaction target. */ +export function resolveCompactionConfiguredTarget(currentModel: Model, availableModels: Model[]): Model | undefined { + return resolveConfiguredModelTarget(currentModel.compactionModel, currentModel, availableModels); +} + +/** Resolves a model role and its explicit thinking selection. */ +export function resolveRoleModelFull( + settings: Settings, + role: string, + availableModels: Model[], + currentModel: Model | undefined, +): ResolvedModelRoleValue { + const roleModelStr = + role === "default" + ? (settings.getModelRole("default") ?? + (currentModel ? `${currentModel.provider}/${currentModel.id}` : undefined)) + : settings.getModelRole(role); + if (!roleModelStr) { + return { model: undefined, thinkingLevel: undefined, explicitThinkingLevel: false, warning: undefined }; + } + return resolveModelRoleValue(roleModelStr, availableModels, { + settings, + matchPreferences: getModelMatchPreferences(settings), + }); +} diff --git a/packages/coding-agent/src/session/session-advisors.ts b/packages/coding-agent/src/session/session-advisors.ts new file mode 100644 index 000000000..8fe37f2dc --- /dev/null +++ b/packages/coding-agent/src/session/session-advisors.ts @@ -0,0 +1,1679 @@ +import { + Agent, + type AgentMessage, + type AgentTool, + AppendOnlyContextManager, + type CompactionSummaryMessage, + countTokens, + resolveTelemetry, + type StreamFn, + ThinkingLevel, +} from "@oh-my-pi/pi-agent-core"; +import { + type CompactionResult, + calculateContextTokens, + compact, + compactionContextTokens, + createCompactionSummaryMessage, + estimateTokens, + prepareCompaction, + type SessionMessageEntry, + shouldCompact, +} from "@oh-my-pi/pi-agent-core/compaction"; +import type { + AssistantMessage, + CodexCompactionContext, + Context, + Message, + Model, + ProviderSessionState, + ServiceTier, + SimpleStreamOptions, +} from "@oh-my-pi/pi-ai"; +import { isUsageLimitOutcome, resolveModelServiceTier } from "@oh-my-pi/pi-ai"; +import * as AIError from "@oh-my-pi/pi-ai/error"; +import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; +import { extractHttpStatusFromError, extractRetryHint, logger } from "@oh-my-pi/pi-utils"; +import { + ADVISOR_DEFAULT_TOOL_NAMES, + AdviseTool, + type AdvisorAgent, + type AdvisorConfig, + AdvisorEmissionGuard, + type AdvisorMessageDetails, + type AdvisorNote, + AdvisorOutputQuarantinedError, + AdvisorRuntime, + type AdvisorRuntimeStatus, + type AdvisorSeverity, + AdvisorTranscriptRecorder, + advisorTranscriptFilename, + annotateForStaleness, + buildAdvisorQuarantineSourceText, + formatAdvisorBatchContent, + getOrCreateAdvisorProviderSessionId, + isAdvisorInterruptImmuneTurnActive, + isInterruptingSeverity, + quarantineAdvisorUnsafeOutput, + resolveAdvisorDeliveryChannel, + slugifyAdvisorName, +} from "../advisor"; +import type { ModelRegistry } from "../config/model-registry"; +import { + formatModelString, + formatModelStringWithRouting, + resolveAdvisorRoleSelection, + resolveModelOverride, +} from "../config/model-resolver"; +import { MODEL_ROLES } from "../config/model-roles"; +import { serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; +import type { Settings } from "../config/settings"; +import { CursorExecHandlers } from "../cursor"; +import { estimateToolSchemaTokens } from "../modes/utils/context-usage"; +import type { PlanModeState } from "../plan-mode/state"; +import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" }; +import type { SecretObfuscator } from "../secrets/obfuscator"; +import { + concreteThinkingLevel, + resolveThinkingLevelForModel, + shouldDisableReasoning, + toReasoningEffort, +} from "../thinking"; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { ClientBridge } from "./client-bridge"; +import type { CustomMessage, CustomMessagePayload } from "./messages"; +import { isAdvisorCard, isTerminalTextAssistantAnswer } from "./queued-messages"; +import { + formatRetryFallbackSelector, + getRetryFallbackRevertPolicy, + parseRetryFallbackSelector, + type RetryFallbackSelector, +} from "./retry-fallback-chains"; +import { formatSessionDumpText } from "./session-dump-format"; +import type { CompactionEntry, SessionEntry } from "./session-entries"; +import { formatSessionHistoryMarkdown } from "./session-history-format"; +import type { SessionManager } from "./session-manager"; +import type { YieldQueue } from "./yield-queue"; + +/** Advisor statistics for the advisor status command. */ +export interface AdvisorStats { + configured: boolean; + active: boolean; + model?: Model; + contextWindow: number; + contextTokens: number; + tokens: { + input: number; + output: number; + reasoning: number; + cacheRead: number; + cacheWrite: number; + total: number; + }; + cost: number; + messages: { + user: number; + assistant: number; + total: number; + }; + /** Per-advisor breakdown for every configured advisor. */ + advisors: PerAdvisorStat[]; +} + +/** One advisor's status, token usage, cost, and message counts. */ +export interface PerAdvisorStat { + name: string; + status: AdvisorRuntimeStatus; + model?: Model; + contextWindow: number; + contextTokens: number; + tokens: AdvisorStats["tokens"]; + cost: number; + messages: AdvisorStats["messages"]; + sessionId?: string; +} + +interface AdvisorRetryFallbackState { + role: string; + originalSelector: string; + originalThinkingLevel: ThinkingLevel; + lastAppliedThinkingLevel: ThinkingLevel; +} + +interface ActiveAdvisor { + name: string; + slug: string; + agent: Agent; + runtime: AdvisorRuntime; + adviseTool: AdviseTool; + emissionGuard: AdvisorEmissionGuard; + recorder: AdvisorTranscriptRecorder; + recorderClosed: Promise; + agentUnsubscribe?: () => void; + model: Model; + thinkingLevel: ThinkingLevel; + providerSessionId: string | undefined; + retryFallback?: AdvisorRetryFallbackState; + retryFallbackPendingSuccess: boolean; + signature: string; +} + +interface AdvisorCompactionSummaryMessage extends CompactionSummaryMessage { + firstKeptEntryId?: string; + advisorUsageAnchorStartIndex?: number; +} + +interface AdvisorRuntimeDescriptor { + config: AdvisorConfig; + name: string; + slug: string; + model: Model; + thinkingLevel: ThinkingLevel; + signature: string; +} + +/** Inputs that configure the advisor roster owned by a session. */ +export interface SessionAdvisorsOptions { + enabled: boolean; + tools?: AgentTool[]; + watchdogPrompt?: string; + sharedInstructions?: string; + contextPrompt?: string; + configs?: AdvisorConfig[]; + streamFn?: StreamFn; + transformProviderContext?: (context: Context, model: Model) => Context | Promise; +} + +/** Options accepted when an advisor injects a primary-session message. */ +export interface AdvisorMessageDeliveryOptions { + triggerTurn?: boolean; + deliverAs?: "steer" | "followUp" | "nextTurn"; + queueChipText?: string; + acceptTerminalEmptyStop?: boolean; +} + +/** Session capabilities borrowed by the advisor controller. */ +export interface SessionAdvisorsHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + modelRegistry: ModelRegistry; + yieldQueue: YieldQueue; + obfuscator: SecretObfuscator | undefined; + providerSessionState: Map; + preferWebsockets: boolean | undefined; + onPayload: SimpleStreamOptions["onPayload"] | undefined; + onResponse: SimpleStreamOptions["onResponse"] | undefined; + onSseEvent: SimpleStreamOptions["onSseEvent"] | undefined; + agentKind(): "main" | "sub"; + isDisposed(): boolean; + abortInProgress(): boolean; + allowAgentInitiatedTurns(): boolean; + planModeState(): PlanModeState | undefined; + clientBridge(): ClientBridge | undefined; + emitSessionEvent(event: AgentSessionEvent): Promise; + emitNotice(level: "info" | "warning" | "error", message: string, source?: string): void; + sendCustomMessage(message: CustomMessagePayload, options?: AdvisorMessageDeliveryOptions): Promise; + extractQueuedAdvisorCards(): CustomMessage[]; + dropPendingAdvisorCards(): void; + preserveAdvisorCard(card: CustomMessage): void; + hasPendingNextTurnMessages(): boolean; + convertToLlmForSideRequest(messages: AgentMessage[]): Message[]; + effectiveServiceTier(model: Model): ServiceTier | undefined; + resolveContextPromotionTarget(currentModel: Model, contextWindow: number): Promise; + resolveCompactionModelCandidates(preferredModel: Model | null | undefined, availableModels: Model[]): Model[]; + resolveRetryFallbackRole(currentSelector: string, currentModel?: Model | null): string | undefined; + findRetryFallbackCandidates( + role: string, + currentSelector: string, + currentModel?: Model | null, + ): RetryFallbackSelector[]; + isRetryFallbackSelectorSuppressed(selector: RetryFallbackSelector): boolean; + noteRetryFallbackCooldown(currentSelector: string, retryAfterMs: number | undefined, errorMessage: string): void; + createCodexCompactionContext(options: { + trigger: CodexCompactionContext["trigger"]; + reason: CodexCompactionContext["reason"]; + phase: CodexCompactionContext["phase"]; + }): CodexCompactionContext; + sessionId(): string; +} + +/** Owns advisor runtimes, delivery policy, context maintenance, and status reporting. */ +export class SessionAdvisors { + readonly #host: SessionAdvisorsHost; + #advisorEnabled: boolean; + #advisorTools: AgentTool[] | undefined; + #advisorWatchdogPrompt: string | undefined; + #advisorSharedInstructions: string | undefined; + #advisorContextPrompt: string | undefined; + #advisorStreamFn: StreamFn | undefined; + #transformProviderContext: ((context: Context, model: Model) => Context | Promise) | undefined; + #advisors: ActiveAdvisor[] = []; + #advisorConfigs: AdvisorConfig[] | undefined; + #advisorStatuses = new Map(); + #advisorProviderSessionIds = new Map(); + #advisorRecorderClosed: Promise = Promise.resolve(); + #advisorAutoResumeSuppressed = false; + #preserveAdvisorAdvice = false; + #advisorPrimaryTurnsCompleted = 0; + #advisorInterruptImmuneTurnStart: number | undefined; + #pendingAdvisorCardEvents = new Set>(); + #advisorYieldQueueUnsubscribe: (() => void) | undefined; + + constructor(host: SessionAdvisorsHost, options: SessionAdvisorsOptions) { + this.#host = host; + this.#advisorEnabled = options.enabled; + this.#advisorTools = options.tools; + this.#advisorWatchdogPrompt = options.watchdogPrompt; + this.#advisorSharedInstructions = options.sharedInstructions; + this.#advisorContextPrompt = options.contextPrompt; + this.#advisorConfigs = options.configs; + this.#advisorStreamFn = options.streamFn; + this.#transformProviderContext = options.transformProviderContext; + if (this.#advisorEnabled) this.#buildAdvisorRuntime(); + } + + /** Delivers one completed primary turn to every live advisor. */ + async onPrimaryTurnEnd( + messages: AgentMessage[], + willContinue: boolean | undefined, + signal?: AbortSignal, + ): Promise { + this.#advisorPrimaryTurnsCompleted++; + for (const advisor of this.#advisors) { + if (advisor.runtime.disposed) continue; + try { + advisor.runtime.onTurnEnd(messages, { willContinue }); + } catch (error) { + logger.warn("advisor onTurnEnd threw; delta dropped", { advisor: advisor.name, err: String(error) }); + } + } + const syncBacklog = this.#host.settings.get("advisor.syncBacklog"); + if (this.#advisors.length === 0 || syncBacklog === "off") return; + const threshold = Number.parseInt(syncBacklog, 10); + await Promise.all(this.#advisors.map(advisor => advisor.runtime.waitForCatchup(30_000, threshold, signal))); + } + + /** Rebuilds live advisors when role assignments alter their resolved runtime inputs. */ + onModelRolesChanged(): void { + if (!this.#advisorEnabled || this.#host.isDisposed()) return; + if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime(); + this.#buildAdvisorRuntime(true); + } + + /** Starts configured advisor runtimes when they are eligible. */ + buildRuntime(seedToCurrent = false): boolean { + return this.#buildAdvisorRuntime(seedToCurrent); + } + + /** Stops every advisor runtime and starts recorder shutdown. */ + stopRuntime(): void { + this.#stopAdvisorRuntime(); + } + + /** Detaches and drains recorder feeds before transcript artifacts are removed. */ + async detachAndCloseRecorders(): Promise { + const closes: Promise[] = []; + for (const advisor of this.#advisors) { + advisor.agentUnsubscribe?.(); + advisor.agentUnsubscribe = undefined; + advisor.recorderClosed = advisor.recorder.close(); + closes.push(advisor.recorderClosed); + } + await Promise.all(closes); + } + + /** Re-primes advisor transcript views across a conversation boundary. */ + resetSessionState(): void { + this.#resetAdvisorSessionState(); + } + + /** Re-primes advisor transcript views after an in-conversation history rewrite. */ + resetAllRuntimes(): void { + this.#resetAllAdvisorRuntimes(); + } + + /** Whether live runtimes still match the resolved advisor configuration. */ + runtimeMatchesCurrentConfig(): boolean { + return this.#advisorRuntimeMatchesCurrentConfig(); + } + + /** Whether concern/blocker delivery is inside the post-interrupt immunity window. */ + isInterruptImmuneTurnActive(): boolean { + return this.#isAdvisorInterruptImmuneTurnActive(); + } + + /** Latest aggregate recorder-close barrier. */ + recorderClosed(): Promise { + return this.#advisorRecorderClosed; + } + + /** Whether a user interrupt currently suppresses advisor-driven auto-resume. */ + get autoResumeSuppressed(): boolean { + return this.#advisorAutoResumeSuppressed; + } + + set autoResumeSuppressed(value: boolean) { + this.#advisorAutoResumeSuppressed = value; + } + + /** Tracks persistence of a visible advisor card emitted outside the primary loop. */ + trackCardEvent(processing: Promise): void { + this.#pendingAdvisorCardEvents.add(processing); + void processing.finally(() => this.#pendingAdvisorCardEvents.delete(processing)).catch(() => {}); + } + + /** Waits for all advisor-card persistence handlers currently in flight. */ + async waitForPendingCardEvents(): Promise { + await Promise.allSettled([...this.#pendingAdvisorCardEvents]); + } + + // Advisor runtime lifecycle + // ------------------------------------------------------------------------- + #advisorImmuneTurnLimit(): number { + const immuneTurns = this.#host.settings.get("advisor.immuneTurns") as number; + if (!Number.isFinite(immuneTurns) || immuneTurns <= 0) return 0; + return Math.trunc(immuneTurns); + } + + #isAdvisorInterruptImmuneTurnActive(): boolean { + return isAdvisorInterruptImmuneTurnActive({ + completedTurns: this.#advisorPrimaryTurnsCompleted, + immuneTurnStart: this.#advisorInterruptImmuneTurnStart, + immuneTurns: this.#advisorImmuneTurnLimit(), + }); + } + + // The next primary turn number starts the immune-turn window. While the + // interrupting steer is still in flight, completedTurns is lower than this + // start, so duplicate concern/blocker advice is also downgraded. + #recordAdvisorInterruptDelivered(): void { + this.#advisorInterruptImmuneTurnStart = this.#advisorPrimaryTurnsCompleted + 1; + } + + /** + * Re-prime the advisor across a conversation boundary: `/new`, `/branch`, + * `/btw`, `/tree`, and session switch/resume. Beyond {@link AdvisorRuntime.reset} + * (which only re-primes the advisor's transcript view and is also fired by + * within-conversation rewrites like compaction/shake/rewind), this clears the + * session-level interrupt latches so the prior conversation's cooldown cannot + * leak into the new one: the post-interrupt immune-turn window + * (`#advisorPrimaryTurnsCompleted`, `#advisorInterruptImmuneTurnStart`) and the + * user-interrupt auto-resume suppression flag. It also drops advisor deliveries + * still queued against the prior conversation — pending asides in the yield + * queue (advisor entries use `skipIdleFlush`, so they linger until the next + * `drainLazy` rather than self-flushing), interrupting cards parked in the + * agent steer/follow-up queues, and preserved cards deferred to the next turn — + * so none of them inject into the new conversation. + */ + #resetAdvisorSessionState(): void { + // Mute the recorder across the re-prime: AdvisorRuntime.reset() aborts the advisor + // loop, and that abort can emit an `aborted` message_end we must not attribute to + // either session's transcript. Detach, reset, then re-attach the live agent's feed. + for (const a of this.#advisors) { + a.agentUnsubscribe?.(); + a.agentUnsubscribe = undefined; + a.runtime.reset(); + a.adviseTool.resetDeliveredNotes(); + a.emissionGuard.reset(); + this.#attachAdvisorRecorderFeed(a); + } + this.#advisorPrimaryTurnsCompleted = 0; + this.#advisorInterruptImmuneTurnStart = undefined; + this.#advisorAutoResumeSuppressed = false; + this.#host.yieldQueue.clear("advisor"); + this.#host.extractQueuedAdvisorCards(); + this.#host.dropPendingAdvisorCards(); + } + + #resolveAdvisorRuntimeDescriptors(emitWarnings: boolean): AdvisorRuntimeDescriptor[] { + const legacy = !this.#advisorConfigs?.length; + const roster: AdvisorConfig[] = legacy ? [{ name: "default" }] : this.#advisorConfigs!; + const descriptors: AdvisorRuntimeDescriptor[] = []; + const usedSlugs = new Set(); + for (const config of roster) { + let slug = legacy ? "" : slugifyAdvisorName(config.name); + if (slug) { + let candidate = slug; + let n = 2; + while (usedSlugs.has(candidate)) candidate = `${slug}-${n++}`; + slug = candidate; + usedSlugs.add(slug); + } + // Per-advisor toggle: skip disabled advisors but keep them in the + // status map so they show `○` rather than disappearing. + if (config.enabled === false) { + this.#advisorStatuses.set(slug, { name: config.name, status: "paused" }); + continue; + } + + // Resolve the advisor's model: an explicit `model` override wins; else the + // `advisor` role chain. A model that fails to resolve skips just this advisor. + let model: Model | undefined; + let thinkingLevel: ThinkingLevel | undefined; + if (config.model) { + const resolved = resolveModelOverride([config.model], this.#host.modelRegistry, this.#host.settings); + model = resolved.model; + thinkingLevel = concreteThinkingLevel(resolved.thinkingLevel); + if (!model) { + this.#advisorStatuses.set(slug, { name: config.name, status: "no_model" }); + if (emitWarnings) { + this.#host.emitNotice( + "warning", + `Advisor "${config.name}": no model matched "${config.model}"`, + "advisor", + ); + } + continue; + } + } else { + const sel = resolveAdvisorRoleSelection(this.#host.settings, this.#host.modelRegistry.getAvailable()); + if (!sel) { + this.#advisorStatuses.set(slug, { name: config.name, status: "no_model" }); + if (emitWarnings) { + logger.debug("advisor enabled but no model assigned to the 'advisor' role; advisor inactive", { + advisor: config.name, + }); + } + continue; + } + model = sel.model; + thinkingLevel = concreteThinkingLevel(sel.thinkingLevel); + } + // Clamp the effort against the resolved model. Historically we defaulted + // to `ThinkingLevel.Medium` unconditionally, which threw at first stream + // on reasoning models that expose no controllable effort surface + // (e.g. `devin-agent`: Cascade routes by sibling model id, not a wire + // param; `getSupportedEfforts` returns `[]`). `resolveThinkingLevelForModel` + // preserves an explicit `off`, clamps a concrete effort into the model's + // supported range, and returns `undefined` for reasoning models without + // controllable efforts — for that case we forward `Inherit` so no effort + // is sent and reasoning stays enabled (matching the `auto`-path fix for + // Devin models via `clampAutoThinkingEffort`). See #4579. + const requestedLevel = thinkingLevel ?? ThinkingLevel.Medium; + const resolvedLevel = resolveThinkingLevelForModel(model, requestedLevel); + const advisorThinkingLevel: ThinkingLevel = resolvedLevel ?? ThinkingLevel.Inherit; + // Record the status entry now (in roster order) so the Map's insertion + // order matches the configured roster even when earlier advisors were + // skipped as paused/no_model. The build loop overwrites this to "running" + // without changing insertion order. + this.#advisorStatuses.set(slug, { name: config.name, status: "running" }); + descriptors.push({ + config, + name: config.name, + slug, + model, + thinkingLevel: advisorThinkingLevel, + signature: this.#advisorRuntimeSignature(config, slug, model, advisorThinkingLevel), + }); + } + return descriptors; + } + + #advisorRuntimeSignature(config: AdvisorConfig, slug: string, model: Model, thinkingLevel: ThinkingLevel): string { + const tools = config.tools?.length ? config.tools.join("\u001e") : ""; + const instructions = config.instructions?.trim() ?? ""; + return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join( + "\u001f", + ); + } + + #advisorRuntimeMatchesCurrentConfig(): boolean { + const descriptors = this.#resolveAdvisorRuntimeDescriptors(false); + if (descriptors.length !== this.#advisors.length) return false; + for (let i = 0; i < descriptors.length; i++) { + if (descriptors[i].signature !== this.#advisors[i].signature) return false; + } + return true; + } + + #buildAdvisorRuntime(seedToCurrent = false): boolean { + if (this.#host.isDisposed()) return false; + if (this.#advisors.length > 0) return true; + if (!this.#advisorEnabled) return false; + if (this.#host.agentKind() !== "main" && !this.#host.settings.get("advisor.subagents")) return false; + + // Rebuild the status map from scratch so removed/renamed advisors don't + // leave stale entries. #resolveAdvisorRuntimeDescriptors populates every + // entry (`paused`/`no_model`/`running`) in roster order; the build loop + // below confirms `running` for successfully built advisors. + this.#advisorStatuses.clear(); + const descriptors = this.#resolveAdvisorRuntimeDescriptors(true); + + // Advisor service tier (`tier.advisor`): "none" (default) runs the advisor + // on standard processing; "inherit" tracks the session's live per-family + // tiers per request (like the main agent, including /fast toggles); a + // concrete value is broadcast across families and applied to the advisor + // model's family. One value for all advisors. + const advisorTierSetting = this.#host.settings.get("tier.advisor"); + const advisorTierMap = + advisorTierSetting === "inherit" + ? undefined + : serviceTierForAllFamilies(serviceTierSettingToTier(advisorTierSetting)); + const advisorServiceTierResolver = (model: Model): ServiceTier | undefined => + advisorTierSetting === "inherit" + ? this.#host.effectiveServiceTier(model) + : resolveModelServiceTier(advisorTierMap, model); + + for (const descriptor of descriptors) { + const { + config, + slug, + model: advisorModel, + name: advisorName, + thinkingLevel: advisorThinkingLevel, + signature, + } = descriptor; + + const emissionGuard = new AdvisorEmissionGuard(); + const adviseTool = new AdviseTool((note, severity) => this.#routeAdvice(advisorRef, note, severity)); + + // `#advisorWatchdogPrompt` already carries WATCHDOG.md + YAML shared + // instructions; `config.instructions` adds this advisor's specialization. + const systemPrompt = [advisorSystemPrompt]; + if (this.#advisorContextPrompt) systemPrompt.push(this.#advisorContextPrompt); + if (this.#advisorWatchdogPrompt) systemPrompt.push(this.#advisorWatchdogPrompt); + if (this.#advisorSharedInstructions) systemPrompt.push(this.#advisorSharedInstructions); + if (config.instructions?.trim()) systemPrompt.push(config.instructions.trim()); + + const names = config.tools === undefined ? ADVISOR_DEFAULT_TOOL_NAMES : new Set(config.tools); + const tools = (this.#advisorTools ?? []).filter(t => names.has(t.name)); + const advisorLoopTools: AgentTool[] = [adviseTool, ...tools]; + const advisorToolMap = new Map>(); + const availableAdvisorToolNames = new Set(); + for (const tool of advisorLoopTools) { + availableAdvisorToolNames.add(tool.name); + advisorToolMap.set(tool.name, tool); + if (tool.customWireName !== undefined) { + availableAdvisorToolNames.add(tool.customWireName); + advisorToolMap.set(tool.customWireName, tool); + } + } + let quarantinedAdvisorOutput: string | undefined; + let currentAdvisorInput = ""; + + const primaryProviderSessionId = this.#host.sessionId(); + const advisorSessionLabel = slug + ? `${primaryProviderSessionId}-advisor-${slug}` + : `${primaryProviderSessionId}-advisor`; + const advisorProviderSessionId = getOrCreateAdvisorProviderSessionId( + this.#advisorProviderSessionIds, + primaryProviderSessionId, + slug, + ); + const appendOnlyContext = new AppendOnlyContextManager(); + + // Thread the primary's telemetry into the advisor loop so the advisor + // model's GenAI spans + usage/cost hooks fire stamped with the local advisor + // identity. `conversationId` is cleared so provider telemetry falls back to + // the UUIDv7 provider session id, not the local `-advisor` label. + const advisorTelemetry = this.#host.agent.telemetry + ? { + ...this.#host.agent.telemetry, + agent: { + id: advisorSessionLabel, + name: slug ? `${MODEL_ROLES.advisor.name}: ${advisorName}` : MODEL_ROLES.advisor.name, + description: formatModelString(advisorModel), + }, + conversationId: undefined, + } + : undefined; + // Mirror the SDK's provider-shaping options (streamFn/onPayload/..., + // providerSessionState, promptCacheKey, transformProviderContext) so each + // advisor's requests cache, route, and obfuscate like the main turn. + // `promptCacheKey` preserves an explicitly pinned provider cache key + // unchanged so tan/shared-session advisor calls read the exact shard the + // parent turn populated. Otherwise the advisor uses its provider UUIDv7 so + // Codex request identity remains UUID-shaped while local labels keep the + // `-advisor` suffix. + const advisorPromptCacheKey = this.#host.agent.promptCacheKey ?? advisorProviderSessionId; + // On the Cursor provider every tool runs server-side and is dispatched + // back through `cursorExecHandlers`; without this bridge the advisor's + // own tools (including the MCP `advise` tool) return `toolNotFound` and + // no advice is ever routed (issue #5680). Mirrors the primary agent's + // bridge (`sdk.ts`), scoped to this advisor's granted tool set. + // Cursor's native `delete` frame removes files directly, bypassing the + // tool map, so gate it on the advisor actually holding a file-mutating + // tool. A default read-only advisor (advise/read/grep/glob) never gets + // to delete workspace files it was never granted (issue #5680 review). + const advisorCanMutateFiles = advisorToolMap.has("write") || advisorToolMap.has("edit"); + if (advisorCanMutateFiles) availableAdvisorToolNames.add("delete"); + const advisorCursorExecHandlers = new CursorExecHandlers({ + cwd: this.#host.sessionManager.getCwd(), + getCwd: () => this.#host.sessionManager.getCwd(), + tools: advisorToolMap, + allowNativeDelete: advisorCanMutateFiles, + }); + const advisorAgent = new Agent({ + initialState: { + systemPrompt, + model: advisorModel, + thinkingLevel: toReasoningEffort(advisorThinkingLevel), + tools: advisorLoopTools, + }, + appendOnlyContext, + sessionId: advisorProviderSessionId, + promptCacheKey: advisorPromptCacheKey, + providerSessionState: this.#host.providerSessionState, + cursorExecHandlers: advisorCursorExecHandlers, + cwdResolver: () => this.#host.sessionManager.getCwd(), + preferWebsockets: this.#host.preferWebsockets, + getApiKey: requestModel => this.#host.modelRegistry.resolver(requestModel, advisorProviderSessionId), + streamFn: this.#advisorStreamFn, + onPayload: this.#host.onPayload, + onResponse: this.#host.onResponse, + onSseEvent: this.#host.onSseEvent, + transformProviderContext: this.#transformProviderContext, + intentTracing: false, + transformAssistantMessage: message => { + quarantinedAdvisorOutput = quarantineAdvisorUnsafeOutput( + message, + availableAdvisorToolNames, + buildAdvisorQuarantineSourceText(currentAdvisorInput, advisorAgent.state.messages), + ); + }, + telemetry: advisorTelemetry, + serviceTier: undefined, + serviceTierResolver: advisorServiceTierResolver, + }); + advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel)); + + const advisorAgentFacade: AdvisorAgent = { + prompt: async input => { + let quarantined: string | undefined; + try { + quarantinedAdvisorOutput = undefined; + currentAdvisorInput = input; + await advisorAgent.prompt(input); + quarantined = quarantinedAdvisorOutput; + } finally { + quarantinedAdvisorOutput = undefined; + currentAdvisorInput = ""; + } + if (quarantined) throw new AdvisorOutputQuarantinedError(quarantined); + }, + abort: reason => advisorAgent.abort(reason), + reset: () => { + advisorAgent.reset(); + appendOnlyContext.log.clear(); + }, + rollbackTo: count => { + // Drop the failed user batch + synthetic assistant-error turn + // `Agent.#runLoop` appended for a turn ending in `stopReason: "error"`. + const messages = advisorAgent.state.messages; + if (count < messages.length) { + messages.length = count; + } + appendOnlyContext.resetSyncCursor(); + advisorAgent.state.error = undefined; + }, + state: advisorAgent.state, + }; + + // Persist this advisor's turns to `/__advisor[.].jsonl` + // (resolved lazily so it follows session switches) for stats attribution + // and Agent Hub observability, without registering it as a peer. + const recorder = new AdvisorTranscriptRecorder( + () => this.#host.sessionManager.getSessionFile(), + () => this.#host.sessionManager.getCwd(), + advisorTranscriptFilename(slug), + // On the advisor on→off→on toggle, wait for the prior recorders' closes + // so two SessionManagers never hold the same file at once. + this.#advisorRecorderClosed, + ); + const runtime = new AdvisorRuntime(advisorAgentFacade, { + snapshotMessages: () => this.#host.agent.state.messages, + enqueueAdvice: (note, severity) => this.#routeAdvice(advisorRef, note, severity), + maintainContext: incomingTokens => this.#maintainAdvisorContext(advisorRef, incomingTokens), + obfuscator: this.#host.obfuscator, + beginAdvisorUpdate: () => advisorRef.emissionGuard.beginUpdate(), + onTurnError: (error, failedMessages) => this.#recoverAdvisorTurn(advisorRef, error, failedMessages), + onTurnSuccess: async () => { + const fallback = advisorRef.retryFallback; + if (!advisorRef.retryFallbackPendingSuccess || !fallback) return; + advisorRef.retryFallbackPendingSuccess = false; + await this.#host.emitSessionEvent({ + type: "retry_fallback_succeeded", + model: formatRetryFallbackSelector(advisorRef.agent.state.model, advisorRef.thinkingLevel), + role: fallback.role, + }); + }, + notifyFailure: error => { + this.#advisorStatuses.set(slug, { name: advisorName, status: "error" }); + const message = error instanceof Error ? error.message : String(error); + this.#host.emitNotice( + "warning", + `Advisor${slug ? ` "${advisorName}"` : ""} unavailable for ${formatModelString(advisorAgent.state.model)}: ${message}`, + "advisor", + ); + }, + notifyQuotaExhausted: () => { + this.#advisorStatuses.set(slug, { name: advisorName, status: "quota_exhausted" }); + this.#host.emitNotice( + "warning", + `Advisor "${advisorName}" quota exhausted — pausing until reset.`, + "advisor", + ); + }, + }); + + const advisorRef: ActiveAdvisor = { + name: advisorName, + slug, + agent: advisorAgent, + runtime, + adviseTool, + emissionGuard, + recorder, + recorderClosed: Promise.resolve(), + model: advisorModel, + thinkingLevel: advisorThinkingLevel, + providerSessionId: advisorProviderSessionId, + retryFallbackPendingSuccess: false, + signature, + }; + this.#attachAdvisorRecorderFeed(advisorRef); + if (seedToCurrent) runtime.seedTo(this.#host.agent.state.messages.length); + this.#advisorStatuses.set(slug, { name: advisorName, status: "running" }); + this.#advisors.push(advisorRef); + } + + // One shared non-blocking aside channel for all advisors; the build callback + // aggregates every advisor's queued nits into one card (each entry already + // carries its own `advisor` name). + if (this.#advisors.length > 0 && !this.#advisorYieldQueueUnsubscribe) { + this.#advisorYieldQueueUnsubscribe = this.#host.yieldQueue.register("advisor", { + build: entries => + entries.length === 0 + ? null + : ({ + role: "custom", + customType: "advisor", + display: true, + attribution: "agent", + timestamp: Date.now(), + content: formatAdvisorBatchContent(entries), + details: { notes: entries } satisfies AdvisorMessageDetails, + } satisfies CustomMessage), + skipIdleFlush: true, + }); + } + + return this.#advisors.length > 0; + } + + /** + * Route one accepted advice note from `advisor` to the primary. Concern and + * blocker interrupt the running agent through the steering channel; once the + * loop has yielded, `triggerTurn` resumes it. After a terminal text answer with + * no queued work, a concern is preserved as a visible advisor card, while a + * blocker wakes the primary to acknowledge work it handed off incorrectly. + * After a deliberate user interrupt auto-resume is suppressed while idle/unwinding + * (the note becomes a preserved card re-entering on resume); a live-streaming turn is + * steered in directly. A plain nit always rides the non-interrupting YieldQueue + * aside. Suppression by the per-advisor emission guard drops the note silently — + * the model still saw `Recorded.`, so it isn't tempted to rephrase the same note + * past the dedupe. + */ + #hasTerminalTextAnswerWithoutQueuedWork(): boolean { + if (this.#host.agent.hasQueuedMessages() || this.#host.hasPendingNextTurnMessages()) return false; + const messages = this.#host.agent.state.messages; + let tail = messages.length - 1; + while (tail >= 0 && isAdvisorCard(messages[tail])) tail--; + return isTerminalTextAssistantAnswer(messages[tail]); + } + + #routeAdvice(advisor: ActiveAdvisor, note: string, severity?: AdvisorSeverity): void { + if (!advisor.emissionGuard.accept(note)) { + logger.debug("advisor advice suppressed by emission guard", { severity, advisor: advisor.name }); + return; + } + // When newer primary turns already arrived while the advisor model was + // processing this batch, the advice was generated without seeing them. + // Append a lightweight staleness caveat so the primary can weigh recency. + const deliveredNote = annotateForStaleness(note, advisor.runtime.hasFreshBacklog); + // The implicit single ("default") advisor stamps no source name, so its + // agent-facing `` bytes stay identical to the pre-multi-advisor path. + const source = advisor.slug ? advisor.name : undefined; + const interrupting = isInterruptingSeverity(severity); + const channel = resolveAdvisorDeliveryChannel({ + severity, + autoResumeSuppressed: this.#advisorAutoResumeSuppressed, + preserveOnly: this.#preserveAdvisorAdvice, + // Key on the live agent-core loop, not session `isStreaming` (which also + // counts `#promptInFlightCount` during post-turn unwind). Only a running + // loop consumes a steer at its next boundary. + streaming: this.#host.agent.state.isStreaming, + aborting: this.#host.abortInProgress(), + terminalAnswerNoQueuedWork: this.#hasTerminalTextAnswerWithoutQueuedWork(), + interruptImmuneTurnActive: interrupting && this.#isAdvisorInterruptImmuneTurnActive(), + }); + if (channel === "aside") { + this.#host.yieldQueue.enqueue("advisor", { note: deliveredNote, severity, advisor: source }); + return; + } + const notes: AdvisorNote[] = [{ note: deliveredNote, severity, advisor: source }]; + const content = formatAdvisorBatchContent(notes); + const details = { notes } satisfies AdvisorMessageDetails; + if (channel === "preserve") { + this.#host.preserveAdvisorCard({ + role: "custom", + customType: "advisor", + content, + display: true, + attribution: "agent", + details, + timestamp: Date.now(), + }); + return; + } + // A steered interrupting note only continues the run when the session can + // actually start (or is already running) a turn. Two idle cases cannot, so + // `sendCustomMessage({ triggerTurn: true })` would silently bury the card in + // `#pendingNextTurnMessages` until the next user prompt — strictly worse than + // the visible preserved card. Preserve instead: + // - Plan mode: only user-driven turns converge on ask/resolve. + // - ACP bridges with `deferAgentInitiatedTurns`: the client cannot show an + // agent-initiated turn as busy, so idle triggers are refused (#5628 review). + const cannotAutoTrigger = + !this.#host.agent.state.isStreaming && + this.#host.clientBridge()?.deferAgentInitiatedTurns === true && + !this.#host.allowAgentInitiatedTurns(); + if (this.#host.planModeState()?.enabled || cannotAutoTrigger) { + this.#host.preserveAdvisorCard({ + role: "custom", + customType: "advisor", + content, + display: true, + attribution: "agent", + details, + timestamp: Date.now(), + }); + return; + } + // Arm the post-interrupt immune window only now that a turn is actually + // being steered/triggered. A merely preserved card never interrupts, so + // arming earlier would downgrade the next `advisor.immuneTurns` worth of + // real concerns/blockers to skip-idle-flush asides (#5628 review). + this.#recordAdvisorInterruptDelivered(); + void this.#host + .sendCustomMessage( + { customType: "advisor", content, display: true, attribution: "agent", details }, + { deliverAs: "steer", triggerTurn: true }, + ) + .catch(err => logger.debug("advisor delivery failed", { err: String(err) })); + } + + /** Re-prime every advisor's transcript view (compaction/shake/rewind) without the + * session-level latch reset {@link #resetAdvisorSessionState} performs. */ + #resetAllAdvisorRuntimes(): void { + for (const a of this.#advisors) a.runtime.reset(); + } + + #stopAdvisorRuntime(): void { + // Detach each recorder feed BEFORE aborting its advisor agent: dispose() aborts + // the loop, and an abort emits a final `message_end` we must not enqueue against + // a closing recorder (it would reopen and resurrect an already-released file). + const closes: Promise[] = []; + for (const a of this.#advisors) { + a.agentUnsubscribe?.(); + a.agentUnsubscribe = undefined; + a.runtime.dispose(); + // Capture each close so dispose()/`/drop` can await the queued open+append+close — + // the last advisor turn would otherwise be lost on a fast process exit. + a.recorderClosed = a.recorder.close(); + closes.push(a.recorderClosed); + } + this.#advisorRecorderClosed = Promise.all(closes).then(() => {}); + this.#advisors = []; + this.#advisorYieldQueueUnsubscribe?.(); + this.#advisorYieldQueueUnsubscribe = undefined; + } + + /** Subscribe the advisor agent's finalized messages into the transcript recorder. + * Idempotent-by-replacement: callers detach the prior feed first. Kept separate + * so the re-prime path can mute the feed across an abort-driven reset. */ + #attachAdvisorRecorderFeed(advisor: ActiveAdvisor): void { + advisor.agentUnsubscribe = advisor.agent.subscribe(event => { + if (event.type === "message_end") advisor.recorder.record(event.message); + }); + } + + /** Switch one advisor model while preserving its context and effort invariants. */ + #setAdvisorModel(advisor: ActiveAdvisor, model: Model, requestedThinkingLevel: ThinkingLevel): ThinkingLevel { + const resolvedThinkingLevel = resolveThinkingLevelForModel(model, requestedThinkingLevel); + const nextThinkingLevel = resolvedThinkingLevel ?? ThinkingLevel.Inherit; + advisor.agent.setModel(model); + advisor.agent.setThinkingLevel(toReasoningEffort(nextThinkingLevel)); + advisor.agent.setDisableReasoning(shouldDisableReasoning(nextThinkingLevel)); + advisor.agent.appendOnlyContext?.invalidateForModelChange(); + advisor.model = model; + advisor.thinkingLevel = nextThinkingLevel; + return nextThinkingLevel; + } + + /** Restore an advisor's configured primary once its fallback cooldown expires. */ + async #maybeRestoreAdvisorRetryFallbackPrimary(advisor: ActiveAdvisor): Promise { + const fallback = advisor.retryFallback; + if (!fallback || getRetryFallbackRevertPolicy(this.#host.settings) !== "cooldown-expiry") return; + + const originalSelector = parseRetryFallbackSelector(fallback.originalSelector, this.#host.modelRegistry); + if (!originalSelector) { + advisor.retryFallback = undefined; + advisor.retryFallbackPendingSuccess = false; + return; + } + const currentSelector = formatRetryFallbackSelector(advisor.agent.state.model, advisor.thinkingLevel); + if (currentSelector === originalSelector.raw) { + if (!this.#host.isRetryFallbackSelectorSuppressed(originalSelector)) { + advisor.retryFallback = undefined; + advisor.retryFallbackPendingSuccess = false; + } + return; + } + if (this.#host.isRetryFallbackSelectorSuppressed(originalSelector)) return; + + const resolvedPrimary = resolveModelOverride( + [originalSelector.raw], + this.#host.modelRegistry, + this.#host.settings, + ); + const primaryModel = + resolvedPrimary.model ?? this.#host.modelRegistry.find(originalSelector.provider, originalSelector.id); + if (!primaryModel) return; + const apiKey = await this.#host.modelRegistry.getApiKey(primaryModel, advisor.providerSessionId); + if (!apiKey) return; + + const thinkingToApply = + advisor.thinkingLevel === fallback.lastAppliedThinkingLevel + ? fallback.originalThinkingLevel + : advisor.thinkingLevel; + this.#setAdvisorModel(advisor, primaryModel, thinkingToApply); + this.#host.settings.getStorage()?.recordModelUsage(formatModelStringWithRouting(primaryModel)); + advisor.retryFallback = undefined; + advisor.retryFallbackPendingSuccess = false; + } + + /** + * Apply the advisor's configured provider-failure fallback chain after + * same-provider credential rotation has no usable sibling. + */ + async #recoverAdvisorTurn( + advisor: ActiveAdvisor, + error: unknown, + failedMessages: readonly AgentMessage[], + ): Promise { + if (error instanceof AdvisorOutputQuarantinedError) return false; + + const failedMessage = failedMessages.findLast( + (message): message is AssistantMessage => message.role === "assistant", + ); + if (failedMessage?.stopReason !== "error") { + // Stream setup can reject before any assistant turn is recorded (e.g. + // an HTTP 429 thrown from prompt()); classify the raw error so a + // structural usage limit still marks the exhausted credential. + const message = error instanceof Error ? error.message : String(error); + if (!AIError.isUsageLimit(error) && !isUsageLimitOutcome(extractHttpStatusFromError(error), message)) { + return false; + } + const currentModel = advisor.agent.state.model; + const outcome = await this.#host.modelRegistry.authStorage.markUsageLimitReached( + currentModel.provider, + advisor.providerSessionId, + { + retryAfterMs: extractRetryHint(undefined, message), + baseUrl: currentModel.baseUrl, + modelId: currentModel.id, + }, + ); + return outcome.switched; + } + if (failedMessage.content.some(block => block.type === "toolCall")) return false; + + const currentModel = advisor.agent.state.model; + const message = failedMessage.errorMessage ?? (error instanceof Error ? error.message : String(error)); + const errorId = AIError.classifyMessage({ + api: currentModel.api, + errorId: failedMessage.errorId, + errorMessage: message, + errorStatus: failedMessage.errorStatus, + }); + if (AIError.is(errorId, AIError.Flag.Abort) || AIError.is(errorId, AIError.Flag.UserInterrupt)) return false; + if (AIError.isContextOverflow(failedMessage, currentModel.contextWindow ?? 0)) return false; + + const currentSelector = formatRetryFallbackSelector(currentModel, advisor.thinkingLevel); + + const retryAfterMs = extractRetryHint(undefined, message); + if ( + AIError.is(errorId, AIError.Flag.UsageLimit) || + isUsageLimitOutcome(extractHttpStatusFromError(error), message) + ) { + const outcome = await this.#host.modelRegistry.authStorage.markUsageLimitReached( + currentModel.provider, + advisor.providerSessionId, + { + retryAfterMs, + baseUrl: currentModel.baseUrl, + modelId: currentModel.id, + }, + ); + if (outcome.switched) return true; + } + + const retrySettings = this.#host.settings.getGroup("retry"); + if (!retrySettings.enabled || !retrySettings.modelFallback) return false; + const role = advisor.retryFallback?.role ?? this.#host.resolveRetryFallbackRole(currentSelector, currentModel); + if (!role || this.#host.findRetryFallbackCandidates(role, currentSelector, currentModel).length === 0) + return false; + + this.#host.noteRetryFallbackCooldown(currentSelector, retryAfterMs, message); + for (const selector of this.#host.findRetryFallbackCandidates(role, currentSelector, currentModel)) { + if (this.#host.isRetryFallbackSelectorSuppressed(selector)) continue; + const resolved = resolveModelOverride([selector.raw], this.#host.modelRegistry, this.#host.settings); + const candidate = resolved.model ?? this.#host.modelRegistry.find(selector.provider, selector.id); + if (!candidate || modelsAreEqual(candidate, currentModel)) continue; + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, advisor.providerSessionId); + if (!apiKey) continue; + + const originalThinkingLevel = advisor.thinkingLevel; + const requestedThinkingLevel = selector.thinkingLevel ?? originalThinkingLevel; + const nextThinkingLevel = this.#setAdvisorModel(advisor, candidate, requestedThinkingLevel); + if (advisor.retryFallback) { + advisor.retryFallback.lastAppliedThinkingLevel = nextThinkingLevel; + } else { + advisor.retryFallback = { + role, + originalSelector: currentSelector, + originalThinkingLevel, + lastAppliedThinkingLevel: nextThinkingLevel, + }; + } + advisor.retryFallbackPendingSuccess = true; + this.#host.settings.getStorage()?.recordModelUsage(formatModelStringWithRouting(candidate)); + await this.#host.emitSessionEvent({ + type: "retry_fallback_applied", + from: currentSelector, + to: selector.raw, + role, + }); + return true; + } + return false; + } + + async #promoteAdvisorContextModel(advisor: ActiveAdvisor, currentModel: Model): Promise { + const promotionSettings = this.#host.settings.getGroup("contextPromotion"); + if (!promotionSettings.enabled) return false; + const contextWindow = currentModel.contextWindow ?? 0; + if (contextWindow <= 0) return false; + const targetModel = await this.#host.resolveContextPromotionTarget(currentModel, contextWindow); + if (!targetModel) return false; + + // Preserve this advisor's own thinking level (a configured `model:...:high` + // keeps its suffix across a promotion); only the model changes. + const advisorThinkingLevel = advisor.thinkingLevel; + try { + this.#setAdvisorModel(advisor, targetModel, advisorThinkingLevel); + logger.debug("Advisor context promotion switched model on overflow", { + advisor: advisor.name, + from: `${currentModel.provider}/${currentModel.id}`, + to: `${targetModel.provider}/${targetModel.id}`, + }); + return true; + } catch (error) { + logger.warn("Advisor context promotion failed", { + advisor: advisor.name, + from: `${currentModel.provider}/${currentModel.id}`, + to: `${targetModel.provider}/${targetModel.id}`, + error: String(error), + }); + return false; + } + } + + async #maintainAdvisorContext(advisor: ActiveAdvisor, incomingTokens: number): Promise { + await this.#maybeRestoreAdvisorRetryFallbackPrimary(advisor); + const agent = advisor.agent; + + const compactionSettings = this.#host.settings.getGroup("compaction"); + if (compactionSettings.strategy === "off") return false; + if (!compactionSettings.enabled) return false; + + const advisorModel = agent.state.model; + const contextWindow = advisorModel.contextWindow ?? 0; + if (contextWindow <= 0) return false; + + const messages = agent.state.messages; + const estimateOptions = { excludeEncryptedReasoning: true } as const; + let storedConversationTokens = 0; + for (const message of messages) { + storedConversationTokens += estimateTokens(message, estimateOptions); + } + // Provider usage (including cache reads and generated output) is the + // trustworthy anchor for accumulated context. Add only the trailing incoming + // delta to that arm. Floor it by a full local estimate — fixed advisor system + // prompt, tool schemas, stored messages, and incoming delta — so provider + // under-reporting or payload transforms cannot suppress maintenance. + const providerContextTokens = this.#estimateAdvisorContextTokens(messages) + incomingTokens; + const localContextTokens = + countTokens(agent.state.systemPrompt) + + estimateToolSchemaTokens(agent.state.tools) + + storedConversationTokens + + incomingTokens; + const contextTokens = compactionContextTokens(providerContextTokens, localContextTokens); + + if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) { + return false; + } + + // 1. Try promotion first + if (await this.#promoteAdvisorContextModel(advisor, advisorModel)) { + // Promotion succeeded, check if new model has enough space + const newModel = agent.state.model; + const newWindow = newModel.contextWindow ?? 0; + if (newWindow > 0) { + const stillNeedsCompaction = shouldCompact(contextTokens, newWindow, compactionSettings); + if (!stillNeedsCompaction) return false; + } + } + + // 2. Run compaction on advisor messages + const pathEntries: SessionEntry[] = messages.map((message, i) => { + const id = `msg-${i}`; + const parentId = i > 0 ? `msg-${i - 1}` : null; + const timestamp = String(message.timestamp || Date.now()); + + if (message.role === "compactionSummary") { + const advisorSummary = message as AdvisorCompactionSummaryMessage; + return { + type: "compaction", + id, + parentId, + timestamp, + summary: message.summary, + shortSummary: message.shortSummary, + firstKeptEntryId: advisorSummary.firstKeptEntryId || `msg-${i + 1}`, + tokensBefore: message.tokensBefore, + } satisfies CompactionEntry; + } + + return { + type: "message", + id, + parentId, + timestamp, + message, + } satisfies SessionMessageEntry; + }); + + const availableModels = this.#host.modelRegistry.getAvailable(); + const candidates = this.#host.resolveCompactionModelCandidates(advisorModel, availableModels); + if (candidates.length === 0) { + // No compaction candidates, fallback to re-prime + return true; + } + const advisorProviderSessionId = getOrCreateAdvisorProviderSessionId( + this.#advisorProviderSessionIds, + this.#host.sessionId(), + advisor.slug, + ); + const preparation = prepareCompaction(pathEntries, compactionSettings, advisorModel); + if (!preparation) { + // Cannot prepare compaction, fallback to re-prime + return true; + } + + const advisorCompactionThinkingLevel: ThinkingLevel | undefined = agent.state.disableReasoning + ? ThinkingLevel.Off + : agent.state.thinkingLevel; + + // Advisor state is in-memory-only, so snapcompact's frame archive has no + // stable SessionEntry preserveData slot to carry across future advisor + // maintenance runs. Use an LLM summary even when the primary session is + // configured for snapcompact. + + let compactResult: CompactionResult | undefined; + let lastError: unknown; + // Instrument the advisor's overflow-compaction one-shot like the primary + // compaction path so the advisor model's maintenance call also emits spans. + const telemetry = resolveTelemetry(agent.telemetry, advisorProviderSessionId); + + const codexCompaction = this.#host.createCodexCompactionContext({ + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + }); + + for (const candidate of candidates) { + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, advisorProviderSessionId); + if (!apiKey) continue; + + try { + compactResult = await compact( + preparation, + candidate, + this.#host.modelRegistry.resolver(candidate, advisorProviderSessionId), + undefined, + undefined, + { + thinkingLevel: advisorCompactionThinkingLevel, + convertToLlm: messages => this.#host.convertToLlmForSideRequest(messages), + telemetry, + tools: agent.state.tools, + sessionId: advisorProviderSessionId, + promptCacheKey: advisorProviderSessionId, + providerSessionState: this.#host.providerSessionState, + codexCompaction, + }, + ); + break; + } catch (error) { + lastError = error; + } + } + + if (!compactResult) { + logger.warn("Advisor compaction failed, falling back to re-prime", { error: String(lastError) }); + return true; + } + + const summary = compactResult.summary; + const shortSummary = compactResult.shortSummary; + const firstKeptEntryId = compactResult.firstKeptEntryId; + const tokensBefore = compactResult.tokensBefore; + + // The retained messages still carry provider usage from before this + // compaction. Record their exact array boundary on the in-memory summary so + // only assistants appended afterward can become the next usage anchor. + const advisorUsageAnchorStartIndex = preparation.recentMessages.length + 1; + const summaryMessage = { + ...createCompactionSummaryMessage(summary, tokensBefore, new Date().toISOString(), shortSummary), + firstKeptEntryId, + advisorUsageAnchorStartIndex, + } satisfies AdvisorCompactionSummaryMessage; + + agent.replaceMessages([summaryMessage, ...preparation.recentMessages]); + return false; + } + /** + * Prevent advisor notes from starting hidden primary turns while a headless + * caller prints and drains the final primary response. + */ + prepareForHeadlessAdvisorDrain(): void { + this.#preserveAdvisorAdvice = true; + } + + async #waitForPendingAdvisorCardEvents(timeoutMs: number): Promise { + const deadline = Date.now() + Math.max(0, timeoutMs); + while (this.#pendingAdvisorCardEvents.size > 0) { + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) return false; + const settled = Promise.allSettled([...this.#pendingAdvisorCardEvents]).then(() => true as const); + const { promise: timedOut, resolve } = Promise.withResolvers(); + const timer = setTimeout(() => resolve(false), remainingMs); + try { + if (!(await Promise.race([settled, timedOut]))) return false; + } finally { + clearTimeout(timer); + } + } + return true; + } + + /** + * Wait for active advisor reviews and their emitted card events before a + * headless caller disposes the session. Returns `false` and logs work disposal + * will abandon when the shared deadline expires or an advisor fails. + */ + async waitForAdvisorCatchup(timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs; + const results = await Promise.all(this.#advisors.map(advisor => advisor.runtime.waitForCatchup(timeoutMs, 1))); + const cardEventsCaughtUp = await this.#waitForPendingAdvisorCardEvents(Math.max(0, deadline - Date.now())); + const abandoned = this.#advisors.filter( + (advisor, index) => results[index] === false && advisor.runtime.backlog > 0, + ); + if (abandoned.length > 0 || !cardEventsCaughtUp) { + logger.warn("advisor shutdown drain incomplete; disposal will abandon reviews or cards", { + timeoutMs, + advisors: abandoned.map(advisor => ({ name: advisor.name, backlog: advisor.runtime.backlog })), + pendingAdvisorCards: this.#pendingAdvisorCardEvents.size, + }); + return false; + } + return true; + } + /** + * Enable or disable the advisor for this session. The setting is overridden for the session, + * and the runtime is started or stopped to match. + * + * @returns true when the advisor is actively running after the call. + */ + setAdvisorEnabled(enabled: boolean): boolean { + this.#advisorEnabled = enabled; + if (enabled) { + if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime(); + return this.#buildAdvisorRuntime(true); + } + this.#stopAdvisorRuntime(); + return false; + } + + /** + * Toggle the advisor setting and start/stop the runtime accordingly. + * + * @returns true when the advisor is actively running after the call. + */ + toggleAdvisorEnabled(): boolean { + return this.setAdvisorEnabled(!this.#advisorEnabled); + } + + /** + * Replace the live advisor roster from an edited `WATCHDOG.yml` (the `/advisor + * configure` save path). Swaps the configs + shared baseline, then rebuilds the + * runtimes in place so the change applies without a restart. When the advisor is + * disabled the new configs are simply stored for the next enable. + * + * @returns the number of advisors active after the rebuild. + */ + applyAdvisorConfigs(advisors: AdvisorConfig[], sharedInstructions: string | undefined): number { + this.#advisorConfigs = advisors; + this.#advisorSharedInstructions = sharedInstructions; + if (!this.#advisorEnabled) return 0; + this.#stopAdvisorRuntime(); + this.#buildAdvisorRuntime(true); + return this.#advisors.length; + } + + /** + * Whether the advisor setting is enabled for this session. + */ + isAdvisorEnabled(): boolean { + return this.#advisorEnabled; + } + + /** + * Whether a live advisor agent is attached to this session. True only when + * `advisor.enabled` is set AND a model resolved for the `advisor` role AND + * the advisor applies to this agent kind — i.e. the actual runtime exists, + * not merely the setting. Drives the status-line badge and `/dump advisor`. + */ + isAdvisorActive(): boolean { + return this.#advisors.length > 0; + } + + /** + * The names of the tools available to advisors this session (the pool a + * `/advisor configure` editor lists). The advisor is a full agent, so this is the + * full built tool set; a tool whose optional factory returns null (e.g. lsp with + * no servers) is absent. + */ + getAdvisorAvailableToolNames(): string[] { + return (this.#advisorTools ?? []).map(tool => tool.name); + } + + /** + * The live advisor `Agent`, or `undefined` when no advisor runtime is + * attached. Surfaced for diagnostics (`/dump advisor` already serializes + * its transcript via {@link formatAdvisorHistoryAsText}) and so callers can + * verify the advisor inherits the session's provider-shaping options + * (`streamFn`, `promptCacheKey`, `providerSessionState`, ...). + */ + getAdvisorAgent(): Agent | undefined { + return this.#advisors[0]?.agent; + } + + /** + * Lightweight advisor status for the status line: returns just the configured + * flag and per-advisor name/status without computing token/cost breakdowns. + * Avoids re-tokenizing the advisor transcript on every render frame. + */ + getAdvisorStatusOverview(): { configured: boolean; advisors: { name: string; status: AdvisorRuntimeStatus }[] } { + // Override stale map entries with live runtime status: failureNotified/quotaExhausted + // clear on reset() but #advisorStatuses lags until the next build. + const liveStatusBySlug = new Map(); + for (const a of this.#advisors) { + liveStatusBySlug.set( + a.slug, + a.runtime.quotaExhausted ? "quota_exhausted" : a.runtime.failureNotified ? "error" : "running", + ); + } + const advisors = [...this.#advisorStatuses.entries()].map(([slug, { name, status }]) => ({ + name, + status: liveStatusBySlug.get(slug) ?? status, + })); + return { configured: this.#advisorEnabled, advisors }; + } + /** + * Return structured advisor stats for the status command and TUI panel. + */ + getAdvisorStats(): AdvisorStats { + const configured = this.#advisorEnabled; + const liveAdvisors = this.#advisors.map(a => this.#computeAdvisorStat(a)); + // Build the complete roster from #advisorStatuses, which already has the + // correct de-duped slugs as keys. Live advisors (from #advisors) carry full + // token/cost data; disabled/no-model/quota-exhausted advisors appear as + // skeleton entries with just name + status so the status line renders a dot. + const liveStatBySlug = new Map(this.#advisors.map((a, i) => [a.slug, liveAdvisors[i]])); + const roster: PerAdvisorStat[] = []; + for (const [slug, entry] of this.#advisorStatuses) { + const live = liveStatBySlug.get(slug); + if (live) { + roster.push(live); + } else { + roster.push({ + name: entry.name, + status: entry.status, + contextWindow: 0, + contextTokens: 0, + tokens: { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + cost: 0, + messages: { user: 0, assistant: 0, total: 0 }, + }); + } + } + const active = liveAdvisors.length > 0; + if (liveAdvisors.length === 0) { + return { + configured, + active, + contextWindow: 0, + contextTokens: 0, + tokens: { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + cost: 0, + messages: { user: 0, assistant: 0, total: 0 }, + advisors: roster, + }; + } + const tokens = { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }; + const messages = { user: 0, assistant: 0, total: 0 }; + let cost = 0; + let contextTokens = 0; + for (const a of liveAdvisors) { + tokens.input += a.tokens.input; + tokens.output += a.tokens.output; + tokens.reasoning += a.tokens.reasoning; + tokens.cacheRead += a.tokens.cacheRead; + tokens.cacheWrite += a.tokens.cacheWrite; + tokens.total += a.tokens.total; + messages.user += a.messages.user; + messages.assistant += a.messages.assistant; + messages.total += a.messages.total; + cost += a.cost; + contextTokens += a.contextTokens; + } + // Single-advisor displays read the top-level model/window directly; surface the + // first advisor's so the legacy status line stays byte-identical. + return { + configured, + active, + model: liveAdvisors[0].model, + contextWindow: liveAdvisors[0].contextWindow, + contextTokens, + tokens, + cost, + messages, + advisors: roster, + }; + } + + /** Compute one advisor's stats slice (tokens, cost, context, message counts). */ + #computeAdvisorStat(advisor: ActiveAdvisor): PerAdvisorStat { + const model = advisor.agent.state.model; + const messages = advisor.agent.state.messages; + const contextTokens = this.#estimateAdvisorContextTokens(messages); + let input = 0; + let output = 0; + let reasoning = 0; + let cacheRead = 0; + let cacheWrite = 0; + let totalTokens = 0; + let cost = 0; + let user = 0; + let assistant = 0; + for (const message of messages) { + if (message.role === "user") user++; + if (message.role === "assistant") { + assistant++; + const assistantMsg = message as AssistantMessage; + input += assistantMsg.usage.input; + output += assistantMsg.usage.output; + reasoning += assistantMsg.usage.reasoningTokens ?? 0; + cacheRead += assistantMsg.usage.cacheRead; + cacheWrite += assistantMsg.usage.cacheWrite; + totalTokens += assistantMsg.usage.totalTokens; + cost += assistantMsg.usage.cost.total; + } + } + return { + name: advisor.name, + status: advisor.runtime.quotaExhausted + ? "quota_exhausted" + : advisor.runtime.failureNotified + ? "error" + : "running", + model, + contextWindow: model.contextWindow ?? 0, + contextTokens, + tokens: { input, output, reasoning, cacheRead, cacheWrite, total: totalTokens }, + cost, + messages: { user, assistant, total: messages.length }, + sessionId: advisor.agent.sessionId, + }; + } + + /** + * Format a concise advisor status line for ACP/text output. + */ + formatAdvisorStatus(): string { + const stats = this.getAdvisorStats(); + if (!stats.active && stats.advisors.length === 0) { + return stats.configured + ? "Advisor setting is enabled, but no model is assigned to the 'advisor' role." + : "Advisor is disabled."; + } + if (stats.advisors.length <= 1) { + const s = stats.advisors[0]; + if (s && s.status === "no_model") { + return stats.configured + ? "Advisor setting is enabled, but no model is assigned to the 'advisor' role." + : "Advisor is disabled."; + } + const contextLine = + s.contextWindow > 0 + ? `Context: ${s.contextTokens.toLocaleString()} / ${s.contextWindow.toLocaleString()} tokens (${Math.round((s.contextTokens / s.contextWindow) * 100)}%)` + : `Context: ${s.contextTokens.toLocaleString()} tokens`; + const spendParts = [`${s.tokens.input.toLocaleString()} input`, `${s.tokens.output.toLocaleString()} output`]; + if (s.tokens.cacheRead > 0) spendParts.push(`${s.tokens.cacheRead.toLocaleString()} cache read`); + if (s.tokens.cacheWrite > 0) spendParts.push(`${s.tokens.cacheWrite.toLocaleString()} cache write`); + const spendLine = `Spend: ${spendParts.join(", ")}, $${s.cost.toFixed(4)}`; + if (!s.model || s.status !== "running") return `Advisor "${s.name}" is ${s.status.replace("_", " ")}.`; + return `Advisor is enabled (${s.model.provider}/${s.model.id}). ${contextLine}. ${spendLine}.`; + } + const lines = [`Advisors enabled (${stats.advisors.length}):`]; + for (const s of stats.advisors) { + const ctx = + s.contextWindow > 0 + ? `${s.contextTokens.toLocaleString()} / ${s.contextWindow.toLocaleString()} (${Math.round((s.contextTokens / s.contextWindow) * 100)}%)` + : `${s.contextTokens.toLocaleString()}`; + lines.push( + ` • ${s.name}${s.model && s.status === "running" ? ` (${s.model.provider}/${s.model.id})` : ` [${s.status}]`} — context ${ctx} tokens, $${s.cost.toFixed(4)}`, + ); + } + lines.push( + `Totals: ${stats.tokens.input.toLocaleString()} input, ${stats.tokens.output.toLocaleString()} output, $${stats.cost.toFixed(4)}.`, + ); + return lines.join("\n"); + } + + /** + * Estimate the advisor's current context tokens. A successful provider usage + * after the latest advisor compaction is ground truth for the prompt plus its + * generated output; only messages after that anchor are estimated. Usage from + * retained pre-compaction messages is stale and must not immediately retrigger + * maintenance on the newly compacted context. + */ + #estimateAdvisorContextTokens(messages: AgentMessage[]): number { + let usageAnchorStartIndex = 0; + for (let i = messages.length - 1; i >= 0; i--) { + const message = messages[i]; + if (message.role !== "compactionSummary") continue; + const advisorSummary = message as AdvisorCompactionSummaryMessage; + // Advisor summaries created before this runtime-only boundary existed have + // no trustworthy way to distinguish retained from newly appended messages. + // Conservatively ignore every current assistant until the next compaction. + usageAnchorStartIndex = advisorSummary.advisorUsageAnchorStartIndex ?? messages.length; + break; + } + + let lastUsageIndex: number | undefined; + let lastUsage: AssistantMessage["usage"] | undefined; + for (let i = messages.length - 1; i >= usageAnchorStartIndex; i--) { + const message = messages[i]; + if (message.role !== "assistant") continue; + const assistant = message as AssistantMessage; + if (assistant.stopReason !== "aborted" && assistant.stopReason !== "error" && assistant.usage) { + lastUsage = assistant.usage; + lastUsageIndex = i; + break; + } + } + + const estimateOptions = { excludeEncryptedReasoning: true } as const; + if (!lastUsage || lastUsageIndex === undefined) { + let estimated = 0; + for (const message of messages) { + estimated += estimateTokens(message, estimateOptions); + } + return estimated; + } + let trailingTokens = 0; + for (let i = lastUsageIndex + 1; i < messages.length; i++) { + trailingTokens += estimateTokens(messages[i], estimateOptions); + } + return calculateContextTokens(lastUsage) + trailingTokens; + } + + /** + * Format the advisor agent's own transcript (its system prompt, config, + * tools, and the markdown deltas it received plus its thinking/advise/read + * calls) as plain text — the advisor-side equivalent of + * {@link formatSessionAsText}. Returns null when no advisor is active. + */ + formatAdvisorHistoryAsText(options?: { compact?: boolean }): string | null { + if (this.#advisors.length === 0) return null; + const dump = (a: ActiveAdvisor): string => + options?.compact + ? formatSessionHistoryMarkdown(a.agent.state.messages) + : formatSessionDumpText({ + messages: a.agent.state.messages, + systemPrompt: a.agent.state.systemPrompt, + model: a.agent.state.model, + thinkingLevel: a.agent.state.thinkingLevel, + tools: a.agent.state.tools, + }); + if (this.#advisors.length === 1) return dump(this.#advisors[0]); + return this.#advisors + .map(a => `### Advisor: ${a.name} (${a.agent.state.model.provider}/${a.agent.state.model.id})\n\n${dump(a)}`) + .join("\n\n"); + } +} diff --git a/packages/coding-agent/src/session/session-handoff.ts b/packages/coding-agent/src/session/session-handoff.ts new file mode 100644 index 000000000..db587126c --- /dev/null +++ b/packages/coding-agent/src/session/session-handoff.ts @@ -0,0 +1,305 @@ +/** Handoff generation and session transition orchestration. */ + +import * as path from "node:path"; +import { + type Agent, + type AgentMessage, + resolveTelemetry, + type StreamFn, + type ThinkingLevel, +} from "@oh-my-pi/pi-agent-core"; +import { generateHandoffFromContext, renderHandoffPrompt } from "@oh-my-pi/pi-agent-core/compaction"; +import type { Message, Model, ServiceTier, SimpleStreamOptions } from "@oh-my-pi/pi-ai"; +import { logger, Snowflake } from "@oh-my-pi/pi-utils"; +import type { ModelRegistry } from "../config/model-registry"; +import type { Settings } from "../config/settings"; +import type { ExtensionRunner, SessionBeforeSwitchResult } from "../extensibility/extensions"; +import { obfuscateProviderContext, type SecretObfuscator } from "../secrets/obfuscator"; +import type { HandoffResult, SessionHandoffOptions } from "./agent-session-types"; +import type { BashSessionTransition } from "./bash-runner"; +import type { SessionContext } from "./session-context"; +import type { SessionManager } from "./session-manager"; + +function createHandoffContext(document: string): string { + return `\n${document}\n\n\nThe above is a handoff document from a previous session. Use this context to continue the work seamlessly.`; +} + +function createHandoffFileName(date = new Date()): string { + const fileTimestamp = date.toISOString().replace(/[:.]/g, "-"); + return `handoff-${fileTimestamp}.md`; +} + +/** Capabilities borrowed from the owning AgentSession. */ +export interface SessionHandoffHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + modelRegistry: ModelRegistry; + extensionRunner: ExtensionRunner | undefined; + sideStreamFn: StreamFn; + obfuscator: SecretObfuscator | undefined; + model(): Model | undefined; + thinkingLevel(): ThinkingLevel | undefined; + sessionId(): string; + sessionFile(): string | undefined; + baseSystemPrompt(): string[]; + assertVibeSessionTransitionAllowed(action: string): void; + setSkipPostTurnMaintenance(timestamp: number | undefined): void; + obfuscateTextForProvider(text: string | undefined): string | undefined; + deobfuscateFromProvider(text: string): string; + convertMessagesToLlm(messages: AgentMessage[], signal?: AbortSignal): Promise; + prepareSimpleStreamOptions(options: SimpleStreamOptions, provider?: string): SimpleStreamOptions; + effectiveServiceTier(model: Model | undefined): ServiceTier | undefined; + flushPendingBash(): Promise; + beginBashSessionTransition(): BashSessionTransition; + markBashSessionTransition(transition: BashSessionTransition): void; + finishBashSessionTransition(transition: BashSessionTransition, success: boolean): void; + cancelOwnAsyncJobs(): void; + clearCheckpointRuntimeState(): void; + clearFreshProviderSessionId(): void; + syncAgentSessionId(): void; + rekeyMemoryForCurrentSessionId(): void; + resetMemoryContextForNewTranscript(): Promise; + clearPendingNextTurnMessages(): void; + resetTodoCycle(): void; + buildDisplaySessionContext(): SessionContext; + resetAdvisorRuntimes(): void; + syncTodoPhasesFromBranch(): void; +} + +/** Generates handoff documents and owns the handoff session transition. */ +export class SessionHandoff { + #handoffAbortController: AbortController | undefined; + readonly #host: SessionHandoffHost; + + constructor(host: SessionHandoffHost) { + this.#host = host; + } + /** + * Cancel in-progress handoff generation. + */ + abortHandoff(): void { + this.#handoffAbortController?.abort(); + } + + /** + * Check if handoff generation is in progress. + */ + get isGeneratingHandoff(): boolean { + return this.#handoffAbortController !== undefined; + } + + /** + * Generate a handoff document with a oneshot LLM call, then start a new session with it. + * + * @param customInstructions Optional focus for the handoff document + * @param options Handoff execution options + * @returns The handoff document text, or undefined if cancelled/failed + */ + async handoff(customInstructions?: string, options?: SessionHandoffOptions): Promise { + this.#host.assertVibeSessionTransitionAllowed("handoff to a new session"); + const entries = this.#host.sessionManager.getBranch(); + const messageCount = entries.filter(e => e.type === "message").length; + + if (messageCount < 2) { + throw new Error("Nothing to hand off (no messages yet)"); + } + + this.#host.setSkipPostTurnMaintenance(undefined); + + this.#handoffAbortController = new AbortController(); + const handoffAbortController = this.#handoffAbortController; + const handoffSignal = handoffAbortController.signal; + const sourceSignal = options?.signal; + const onSourceAbort = () => { + if (!handoffSignal.aborted) { + handoffAbortController.abort(); + } + }; + if (sourceSignal) { + sourceSignal.addEventListener("abort", onSourceAbort, { once: true }); + if (sourceSignal.aborted) { + onSourceAbort(); + } + } + + try { + if (handoffSignal.aborted) { + throw new Error("Handoff cancelled"); + } + + const model = this.#host.model(); + if (!model) { + throw new Error("No model selected for handoff"); + } + const apiKey = await this.#host.modelRegistry.getApiKey(model, this.#host.sessionId()); + if (!apiKey) { + throw new Error(`No API key for ${model.provider}`); + } + + // Build the handoff request through the SAME pipeline a live turn uses + // (`runEphemeralTurn` / `/btw` share it) so the oneshot reads the + // provider prompt cache the main turn populated instead of cold-missing + // the whole prefix: identical system prompt, normalized tools, and + // transform-/obfuscation-matched message history via + // `convertMessagesToLlm` + `buildSideRequestContext`, plus the live turn's + // effective provider cache key with a unique side `sessionId` so + // OpenAI/Codex append-only state never mixes with the live turn. + const cacheSessionId = this.#host.sessionId(); + // The loop sends `promptCacheKey` (providerPromptCacheKey) and falls back to + // the provider session id; providers route on `promptCacheKey ?? sessionId`. + // Both can diverge from this.#host.sessionId() (tan/subagent/shared sessions), so + // mirror exactly what the live turn populated the cache under. + const handoffPromptCacheKey = this.#host.agent.promptCacheKey ?? this.#host.agent.sessionId; + const handoffPromptText = renderHandoffPrompt(this.#host.obfuscateTextForProvider(customInstructions)); + const handoffSnapshot: AgentMessage[] = [ + ...this.#host.agent.state.messages, + { + role: "user", + content: [{ type: "text", text: handoffPromptText }], + attribution: "agent", + timestamp: Date.now(), + }, + ]; + const handoffLlmMessages = await this.#host.convertMessagesToLlm(handoffSnapshot, handoffSignal); + // Base system prompt, not a per-turn `before_agent_start` hook override — + // the handoff seeds a fresh session and must not carry prompt-specific + // hook state. Matches the prompt the old handoff path sent. + const handoffContext = await this.#host.agent.buildSideRequestContext( + handoffLlmMessages, + this.#host.baseSystemPrompt(), + ); + const handoffStreamOptions = this.#host.prepareSimpleStreamOptions( + { + apiKey: this.#host.modelRegistry.resolver(model, cacheSessionId), + sessionId: `${cacheSessionId}:side:${Snowflake.next()}`, + promptCacheKey: handoffPromptCacheKey, + preferWebsockets: false, + serviceTier: this.#host.effectiveServiceTier(model), + hideThinkingSummary: this.#host.agent.hideThinkingSummary, + initiatorOverride: "agent", + signal: handoffSignal, + }, + model.provider, + ); + const rawHandoffText = await generateHandoffFromContext( + obfuscateProviderContext(this.#host.obfuscator, handoffContext), + model, + { + streamOptions: handoffStreamOptions, + completeImpl: async (requestModel, requestContext, requestOptions) => { + const stream = await this.#host.sideStreamFn(requestModel, requestContext, requestOptions); + return stream.result(); + }, + telemetry: resolveTelemetry(this.#host.agent.telemetry, this.#host.sessionId()), + // Honor the user's /model thinking selection on the handoff path. + // Clamped per-model inside generateHandoffFromContext via + // resolveCompactionEffort so unsupported-effort models don't trip + // requireSupportedEffort. + thinkingLevel: this.#host.thinkingLevel(), + }, + ); + const handoffText = this.#host.deobfuscateFromProvider(rawHandoffText); + + if (handoffSignal.aborted) { + throw new Error("Handoff cancelled"); + } + if (!handoffText) { + return undefined; + } + + // Start a new session + const previousSessionFile = this.#host.sessionFile(); + if (this.#host.extensionRunner?.hasHandlers("session_before_switch")) { + const result = (await this.#host.extensionRunner.emit({ + type: "session_before_switch", + reason: "handoff", + })) as SessionBeforeSwitchResult | undefined; + + if (result?.cancel) { + options?.onSwitchCancelled?.(); + return undefined; + } + } + await this.#host.flushPendingBash(); + await this.#host.sessionManager.flush(); + const bashTransition = this.#host.beginBashSessionTransition(); + this.#host.cancelOwnAsyncJobs(); + let sessionTransitioned = false; + try { + await this.#host.sessionManager.newSession( + previousSessionFile ? { parentSession: previousSessionFile } : undefined, + ); + this.#host.markBashSessionTransition(bashTransition); + sessionTransitioned = true; + } finally { + this.#host.finishBashSessionTransition(bashTransition, sessionTransitioned); + } + + this.#host.clearCheckpointRuntimeState(); + // agent.reset() clears the core steering/follow-up queues. Preserve any queued + // steers/follow-ups (RPC/SDK steer()/followUp() issued during the handoff, or a + // pre-loader TUI steer) so they survive into the post-handoff session instead of + // being silently dropped. Capture is synchronous immediately before reset and + // restore is synchronous immediately after — no await gap — so a steer arriving + // later (during ensureOnDisk/Bun.write below) appends to the restored queue + // rather than being clobbered. + const preservedSteering = this.#host.agent.peekSteeringQueue().slice(); + const preservedFollowUp = this.#host.agent.peekFollowUpQueue().slice(); + this.#host.agent.reset(); + this.#host.agent.replaceQueues(preservedSteering, preservedFollowUp); + this.#host.clearFreshProviderSessionId(); + this.#host.syncAgentSessionId(); + this.#host.rekeyMemoryForCurrentSessionId(); + await this.#host.resetMemoryContextForNewTranscript(); + this.#host.clearPendingNextTurnMessages(); + this.#host.resetTodoCycle(); + + // Inject the handoff document as a custom message + const handoffContent = createHandoffContext(handoffText); + this.#host.sessionManager.appendCustomMessageEntry("handoff", handoffContent, true, undefined, "agent"); + await this.#host.sessionManager.ensureOnDisk(); + let savedPath: string | undefined; + if (options?.autoTriggered && this.#host.settings.get("compaction.handoffSaveToDisk")) { + const artifactsDir = this.#host.sessionManager.getArtifactsDir(); + if (artifactsDir) { + const handoffFilePath = path.join(artifactsDir, createHandoffFileName()); + try { + await Bun.write(handoffFilePath, `${handoffText}\n`); + savedPath = handoffFilePath; + } catch (error) { + logger.warn("Failed to save handoff document to disk", { + path: handoffFilePath, + error: error instanceof Error ? error.message : String(error), + }); + } + } else { + logger.debug("Skipping handoff document save because session is not persisted"); + } + } + + // Rebuild agent messages from session + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.resetAdvisorRuntimes(); + this.#host.syncTodoPhasesFromBranch(); + if (this.#host.extensionRunner) { + await this.#host.extensionRunner.emit({ + type: "session_switch", + reason: "handoff", + previousSessionFile, + }); + } + + return { document: handoffText, savedPath }; + } catch (error) { + if (handoffSignal.aborted || (error instanceof Error && error.name === "AbortError")) { + throw new Error("Handoff cancelled"); + } + throw error; + } finally { + sourceSignal?.removeEventListener("abort", onSourceAbort); + this.#handoffAbortController = undefined; + } + } +} diff --git a/packages/coding-agent/src/session/session-maintenance.ts b/packages/coding-agent/src/session/session-maintenance.ts new file mode 100644 index 000000000..d282b4b68 --- /dev/null +++ b/packages/coding-agent/src/session/session-maintenance.ts @@ -0,0 +1,2983 @@ +/** Context maintenance for an active coding-agent session. */ + +import { scheduler } from "node:timers/promises"; +import { + type Agent, + type AgentMessage, + type AgentTurnEndContext, + countTokens, + resolveTelemetry, + type StreamFn, + type ThinkingLevel, +} from "@oh-my-pi/pi-agent-core"; +import { + AGGRESSIVE_SHAKE_CONFIG, + AUTO_HANDOFF_THRESHOLD_FOCUS, + applyShakeRegions, + CompactionCancelledError, + type CompactionPreparation, + type CompactionResult, + type CompactionSettings, + calculateContextTokens, + collectShakeRegions, + compact, + compactionContextTokens, + createCompactionSummaryMessage, + DEFAULT_SHAKE_CONFIG, + effectiveReserveTokens, + estimateTokens, + prepareCompaction, + resolveBudgetReserveTokens, + resolveThresholdTokens, + type ShakeConfig, + type ShakeRegion, + type SummaryOptions, + shouldCompact, + shouldUseOpenAiRemoteCompaction, +} from "@oh-my-pi/pi-agent-core/compaction"; +import { + DEFAULT_PRUNE_CONFIG, + pruneSupersededToolResults, + pruneToolOutputs, + readToolSupersedeKey, +} from "@oh-my-pi/pi-agent-core/compaction/pruning"; +import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection"; +import type { AssistantMessage, CodexCompactionContext, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai"; +import * as AIError from "@oh-my-pi/pi-ai/error"; +import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; +import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; +import { logger } from "@oh-my-pi/pi-utils"; +import * as snapcompact from "@oh-my-pi/snapcompact"; +import type { ModelRegistry } from "../config/model-registry"; +import { MODEL_ROLE_IDS } from "../config/model-roles"; +import type { Settings } from "../config/settings"; +import { getDefault } from "../config/settings"; +import type { ExtensionRunner, SessionBeforeCompactResult } from "../extensibility/extensions"; +import type { CompactOptions, ContextUsage } from "../extensibility/extensions/types"; +import type { GoalModeState } from "../goals/state"; +import { resolveMemoryBackend } from "../memory-backend/resolve"; +import type { MemoryBackendOperationContext } from "../memory-backend/types"; +import type { NonMessageTokenSource } from "../modes/utils/context-usage"; +import { computeNonMessageTokens } from "../modes/utils/context-usage"; +import { createPlanReadMatcher } from "../plan-mode/plan-protection"; +import type { ConfiguredThinkingLevel } from "../thinking"; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { ContextUsageBreakdown, HandoffResult, SessionHandoffOptions } from "./agent-session-types"; +import { findCompactMode } from "./compact-modes"; +import { convertToLlm, stripImagesFromMessage } from "./messages"; +import { isTerminalTextAssistantAnswer } from "./queued-messages"; +import { + resolveCompactionConfiguredTarget, + resolveContextPromotionConfiguredTarget, + resolveRoleModelFull, +} from "./role-models"; +import type { SessionContext } from "./session-context"; +import { getLatestCompactionEntry } from "./session-context"; +import type { CompactionEntry, SessionEntry } from "./session-entries"; +import type { SessionManager } from "./session-manager"; +import type { ShakeMode, ShakeResult } from "./shake-types"; + +export type CompactionCheckResult = Readonly<{ + deferredHandoff: boolean; + continuationScheduled: boolean; + automaticContinuationBlocked?: boolean; + historyRewritten?: boolean; +}>; + +/** Shared no-op result for dispatcher paths that perform no maintenance. */ +export const COMPACTION_CHECK_NONE: CompactionCheckResult = { + deferredHandoff: false, + continuationScheduled: false, +}; +const COMPACTION_CHECK_DEFERRED_HANDOFF: CompactionCheckResult = { + deferredHandoff: true, + continuationScheduled: false, +}; +const COMPACTION_CHECK_CONTINUATION: CompactionCheckResult = { + deferredHandoff: false, + continuationScheduled: true, +}; +const COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION: CompactionCheckResult = { + deferredHandoff: false, + continuationScheduled: false, + automaticContinuationBlocked: true, +}; + +/** + * User-facing notice for a compaction dead end: maintenance freed too little + * to retry safely. `remedies` names the recovery actions left on the emitting + * path — by the time the post-pass dead end fires, the tiered rescue has + * already attempted both elide and image-drop automatically. + */ +function compactionDeadEndWarning(remedies: string): string { + return ( + "Compaction freed too little context to make progress — pausing automatic maintenance to avoid a compaction loop. " + + `The most recent turn alone is too large to reduce further; ${remedies} or switch to a larger-context model.` + ); +} + +/** Creates one provider-scoped compaction lifecycle descriptor. */ +export function createCodexCompactionContext(options: { + trigger: CodexCompactionContext["trigger"]; + reason: CodexCompactionContext["reason"]; + phase: CodexCompactionContext["phase"]; +}): CodexCompactionContext { + return { + operationId: crypto.randomUUID(), + trigger: options.trigger, + reason: options.reason, + phase: options.phase, + strategy: "memento", + }; +} + +/** + * Per-turn prune cache window. A tool result whose all-message suffix exceeds + * this is in the warm, already-sent prompt-cache prefix: re-writing it costs the + * cacheWrite premium on the whole suffix. Per-turn passes only reclaim inside + * this tail (matches the supersede pass's default `suffixTokenLimit`); deeper + * stale/age victims are left to compaction/shake, which rebuild the cache anyway. + */ +const PRUNE_CACHE_WARM_SUFFIX_TOKENS = 8_000; + +/** + * Idle gap after which the supersede pass may flush the whole sent region (the + * provider cache is cold, so re-writing it is free). MUST exceed the maximum + * Anthropic prompt-cache TTL — "long" retention (the OAuth default) is 1h — or a + * still-warm prefix is busted by the flush. 90 min leaves margin over the 1h TTL. + */ +const PRUNE_IDLE_FLUSH_MS = 90 * 60_000; + +/** + * Hysteresis band for the post-maintenance "did we actually create headroom?" + * check shared by the shake tail and the context-full / snapcompact tail. A + * pass counts as having resolved threshold pressure only when residual context + * lands at or below `COMPACTION_RECOVERY_BAND × threshold`. Re-checking against + * the raw threshold lets a pass keep reclaiming a trickle of the previous + * turn's output and land just under the line every turn, sustaining the + * auto-continue dead loop reported in #2275; the same band stops the + * context-full / snapcompact tail from re-firing on a history whose single + * most-recent kept turn already exceeds the threshold (the snapcompact thrash). + */ +const COMPACTION_RECOVERY_BAND = 0.8; + +function mergeLlmCompactionPreserveData( + hookPreserveData: Record | undefined, + resultPreserveData: Record | undefined, +): Record | undefined { + const preserveData = { ...(hookPreserveData ?? {}), ...(resultPreserveData ?? {}) }; + return snapcompact.stripPreservedArchive(Object.keys(preserveData).length > 0 ? preserveData : undefined); +} + +/** Capabilities borrowed from the owning AgentSession. */ +export interface SessionMaintenanceHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + modelRegistry: ModelRegistry; + extensionRunner: ExtensionRunner | undefined; + sideStreamFn: StreamFn; + providerSessionState: Map; + model(): Model | undefined; + thinkingLevel(): ThinkingLevel | undefined; + isDisposed(): boolean; + isStreaming(): boolean; + isGeneratingHandoff(): boolean; + promptGeneration(): number; + sessionId(): string; + messages(): AgentMessage[]; + baseSystemPrompt(): string[]; + goalModeState(): GoalModeState | undefined; + planReferencePath(): string; + nonMessageTokenSource(): NonMessageTokenSource; + memoryBackendSession(): MemoryBackendOperationContext["session"]; + emitSessionEvent(event: AgentSessionEvent): Promise; + emitNotice(level: "info" | "warning" | "error", message: string, source?: string): void; + schedulePostPromptTask( + task: (signal: AbortSignal) => Promise, + options?: { delayMs?: number; generation?: number; onSkip?: (reason: "aborted" | "stale-generation") => void }, + ): void; + scheduleAgentContinue(options?: { + delayMs?: number; + generation?: number; + shouldContinue?: () => boolean; + onSkip?: ( + reason: + | "aborted" + | "stale-generation" + | "session-unavailable" + | "should-continue-false" + | "post-restore-unavailable", + ) => void; + onError?: () => void; + }): void; + scheduleCompactionContinuation(options: { + generation: number; + autoContinue: boolean; + terminalTextAnswer: boolean; + suppressContinuation: boolean; + }): boolean; + persistTurnMessagesForMidRunCompaction(context: AgentTurnEndContext | undefined): Promise; + findLastAssistantMessage(): AssistantMessage | undefined; + disconnectFromAgent(): void; + reconnectToAgent(): void; + drainStrandedQueuedMessages(): void; + buildDisplaySessionContext(): SessionContext; + convertToLlmForSideRequest(messages: AgentMessage[]): Message[]; + obfuscateTextForProvider(text: string | undefined): string | undefined; + obfuscatePreparationForProvider(preparation: CompactionPreparation): CompactionPreparation; + closeCodexProviderSessionsForHistoryRewrite(): void; + resetCodexProviderAfterCompaction(compaction: CodexCompactionContext): void; + resetPlanReference(): void; + syncTodoPhasesFromBranch(): void; + resetAdvisorRuntimes(): void; + rebaseAfterCompaction(): void; + getContextBreakdown(options?: { + contextWindow?: number; + pendingMessages?: AgentMessage[]; + }): ContextUsageBreakdown | undefined; + getContextUsage(options?: { contextWindow?: number }): ContextUsage | undefined; + shake(mode: ShakeMode, options?: { config?: ShakeConfig; signal?: AbortSignal }): Promise; + dropImages(): Promise<{ removed: number }>; + runHandoff(customInstructions?: string, options?: SessionHandoffOptions): Promise; + removeAssistantMessageFromActiveContext(message: AssistantMessage): void; + dropPersistedAssistantTurn(message: AssistantMessage): Promise; + runRecoveryCompactionWithRollback( + reason: "overflow" | "incomplete", + message: AssistantMessage, + allowDefer: boolean, + options: { autoContinue: boolean; triggerContextTokens?: number }, + ): Promise; + parseRetryAfterMsFromError(errorMessage: string): number | undefined; + setModelTemporary( + model: Model, + thinkingLevel?: ConfiguredThinkingLevel, + options?: { ephemeral?: boolean }, + ): Promise; + abort(options?: { + goalReason?: "interrupted" | "internal"; + reason?: string; + preserveCompaction?: boolean; + }): Promise; + abortHandoff(): void; +} + +/** Owns compaction, pruning, shake, promotion, and automatic context maintenance. */ +export class SessionMaintenance { + #compactionAbortController: AbortController | undefined; + #autoCompactionAbortController: AbortController | undefined; + #skipPostTurnMaintenanceAssistantTimestamp: number | undefined; + readonly #host: SessionMaintenanceHost; + + get #model(): Model | undefined { + return this.#host.model(); + } + + get #goalModeState(): GoalModeState | undefined { + return this.#host.goalModeState(); + } + + constructor(host: SessionMaintenanceHost) { + this.#host = host; + } + + /** Whether manual or automatic context maintenance is active. */ + get isCompacting(): boolean { + return this.#autoCompactionAbortController !== undefined || this.#compactionAbortController !== undefined; + } + + /** Assistant timestamp whose post-turn maintenance must be skipped once. */ + get skipPostTurnMaintenanceAssistantTimestamp(): number | undefined { + return this.#skipPostTurnMaintenanceAssistantTimestamp; + } + + set skipPostTurnMaintenanceAssistantTimestamp(timestamp: number | undefined) { + this.#skipPostTurnMaintenanceAssistantTimestamp = timestamp; + } + /** + * Append plan-read protection to a prune/shake config so the active plan + * file survives compaction alongside skill reads (the config defaults + * already carry skill protection). The matcher reads the current plan + * reference path at match time, so retitled plans are covered. + */ + #withPlanProtection(config: T): T { + const planMatcher = createPlanReadMatcher(() => this.#host.planReferencePath()); + return { ...config, protectedTools: [...config.protectedTools, planMatcher] }; + } + + async #pruneToolOutputs(): Promise<{ prunedCount: number; tokensSaved: number } | undefined> { + const branchEntries = this.#host.sessionManager.getBranch(); + const keepBoundaryId = getLatestCompactionEntry(branchEntries)?.firstKeptEntryId; + const result = pruneToolOutputs( + branchEntries, + this.#withPlanProtection({ + ...DEFAULT_PRUNE_CONFIG, + pruneUseless: this.#host.settings.getGroup("compaction").dropUseless, + // Cache-stable boundary: never re-write the warm, already-sent prefix + // (deep stale/age victims) or summarized-away entries every turn. + keepBoundaryId, + cacheWarmSuffixTokens: PRUNE_CACHE_WARM_SUFFIX_TOKENS, + }), + ); + if (result.prunedCount === 0) { + return undefined; + } + + await this.#host.sessionManager.rewriteEntries(); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.resetAdvisorRuntimes(); + this.#host.syncTodoPhasesFromBranch(); + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + return result; + } + + /** + * Per-turn stale-result pass: prune older `read` results that a newer read + * of the same file has made stale, plus results their tool flagged + * contextually useless. Cache-aware (only fires when the suffix after a + * candidate is small or the session has been idle long enough that the + * provider prompt cache is cold), so it is cheap to run every turn. Gated + * on the `compaction.supersedeReads` and `compaction.dropUseless` settings. + * + * Persists via `rewriteEntries` like every other history rewrite — the + * session file must match the live (pruned) context or file-based forks + * (`/fork`, `/tan`) and resume rebuild a divergent prefix and cold-miss the + * provider prompt cache. + */ + async #pruneStaleToolResults(): Promise<{ prunedCount: number; tokensSaved: number } | undefined> { + const { supersedeReads, dropUseless } = this.#host.settings.getGroup("compaction"); + if (!supersedeReads && !dropUseless) return undefined; + const branchEntries = this.#host.sessionManager.getBranch(); + const keepBoundaryId = getLatestCompactionEntry(branchEntries)?.firstKeptEntryId; + const result = pruneSupersededToolResults( + branchEntries, + this.#withPlanProtection({ + supersedeKey: supersedeReads ? readToolSupersedeKey : undefined, + pruneUseless: dropUseless, + protectedTools: [...DEFAULT_PRUNE_CONFIG.protectedTools], + // Never re-write summarized-away entries; only flush the whole sent + // region once the cache is genuinely cold (idle exceeds the 1h TTL). + keepBoundaryId, + idleFlushMs: PRUNE_IDLE_FLUSH_MS, + }), + ); + if (result.prunedCount === 0) { + return undefined; + } + + await this.#host.sessionManager.rewriteEntries(); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.resetAdvisorRuntimes(); + this.#host.syncTodoPhasesFromBranch(); + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + return result; + } + + /** + * Strip image content blocks from every message on the current branch and + * persist the rewrite. Walks `SessionManager.getBranch()` in place — both + * `SessionMessageEntry.message` and `CustomMessageEntry.content` arrays + * are mutated, then `rewriteEntries` durably commits the new shape. The + * agent's runtime view is rebuilt from the freshly-mutated entries so any + * provider sessions caching message identity (Codex Responses) are torn + * down to force a clean replay on the next turn. + * + * No-op when the branch carries no images; returns `{ removed: 0 }` and + * skips the disk rewrite. + */ + async dropImages(): Promise<{ removed: number }> { + const branchEntries = this.#host.sessionManager.getBranch(); + let removed = 0; + for (const entry of branchEntries) { + if (entry.type === "message") { + removed += stripImagesFromMessage(entry.message); + continue; + } + if (entry.type === "custom_message" && typeof entry.content !== "string") { + const kept: typeof entry.content = []; + let dropped = 0; + for (const part of entry.content) { + if (part.type === "image") { + dropped++; + } else { + kept.push(part); + } + } + if (dropped > 0) { + if (kept.length === 0) { + kept.push({ type: "text", text: "[image removed]" }); + } + entry.content = kept; + removed += dropped; + } + } + } + if (removed === 0) { + return { removed: 0 }; + } + await this.#host.sessionManager.rewriteEntries(); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.resetAdvisorRuntimes(); + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + return { removed }; + } + + /** + * Surgically reduce context by dropping heavy content ("shake"). + * + * - `images` delegates to {@link dropImages}. + * - `elide` replaces whole tool-call results and large fenced/XML blocks + * with short placeholders that embed an `artifact://` recovery link. + * + * Mutates the branch in place, persists via `rewriteEntries`, replays the + * rebuilt context through the agent, and tears down provider sessions that + * cache message identity — same rewrite contract as {@link dropImages}. + * + * No-op (zero counts) when nothing is eligible. + */ + async shake(mode: ShakeMode, opts: { config?: ShakeConfig; signal?: AbortSignal } = {}): Promise { + if (mode === "images") { + const { removed } = await this.#host.dropImages(); + return { mode, toolResultsDropped: 0, blocksDropped: 0, imagesDropped: removed, tokensFreed: 0 }; + } + + const branchEntries = this.#host.sessionManager.getBranch(); + const config = this.#withPlanProtection({ + ...(opts.config ?? AGGRESSIVE_SHAKE_CONFIG), + // Skip entries summarized away by the latest compaction — shaking them + // only churns persisted history with no prompt/cache effect. + keepBoundaryId: getLatestCompactionEntry(branchEntries)?.firstKeptEntryId, + }); + const regions = collectShakeRegions(branchEntries, config); + if (regions.length === 0) { + return { mode, toolResultsDropped: 0, blocksDropped: 0, tokensFreed: 0 }; + } + + const artifactId = await this.#saveShakeArtifact(regions); + const replacements = regions.map((region, index) => this.#shakeElidePlaceholder(region, index, artifactId)); + + let toolResultsDropped = 0; + let blocksDropped = 0; + let originalTokens = 0; + let replacementTokens = 0; + const items = regions.map((region, index) => { + if (region.kind === "toolResult") toolResultsDropped++; + else blocksDropped++; + originalTokens += region.tokens; + const replacement = replacements[index]; + if (replacement.length > 0) replacementTokens += countTokens(replacement); + return { region, replacement }; + }); + + applyShakeRegions(items); + + await this.#host.sessionManager.rewriteEntries(); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.resetAdvisorRuntimes(); + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + + return { + mode, + toolResultsDropped, + blocksDropped, + tokensFreed: Math.max(0, originalTokens - replacementTokens), + artifactId, + }; + } + + #shakeElidePlaceholder(region: ShakeRegion, index: number, artifactId: string | undefined): string { + if (artifactId) { + return `[shaken ~${region.tokens} tokens — recover: artifact://${artifactId} (region ${index + 1})]`; + } + return `[shaken ~${region.tokens} tokens]`; + } + + /** + * Concatenate the original region contents into one session artifact so the + * agent can read them back via `artifact://`. Returns `undefined` when + * the session is not persisted or the write fails — callers degrade to a + * bare placeholder. + */ + async #saveShakeArtifact(regions: ShakeRegion[]): Promise { + const parts: string[] = []; + for (let i = 0; i < regions.length; i++) { + const region = regions[i]; + parts.push(`### region ${i + 1} (${region.label}, ~${region.tokens} tok)`, "", region.originalText, ""); + } + try { + return await this.#host.sessionManager.saveArtifact(parts.join("\n"), "shake"); + } catch { + return undefined; + } + } + + /** + * Manually compact the session context. + * Aborts current agent operation first. + * @param customInstructions Optional instructions for the compaction summary + * @param options Optional callbacks for completion/error handling + */ + async compact(customInstructions?: string, options?: CompactOptions): Promise { + if (this.#compactionAbortController) { + throw new Error("Compaction already in progress"); + } + // Resolve the `/compact ` subcommand up front so input validation + // runs before we disconnect/abort the active agent operation below. + const compactMode = options?.mode ? findCompactMode(options.mode) : undefined; + // Modes that produce no LLM summary (snapcompact) have nothing to focus. + // Reject focus text loudly so programmatic callers don't silently lose + // instructions (the slash path pre-validates via parseCompactArgs). + // `internalGuidance` counts the same way — plan-mode approval never + // combines with a rejects-focus mode, but reject early if a caller ever + // wires it up so we don't silently drop the directive on the snapcompact + // fallback (issue #4359). + if (compactMode?.rejectsFocus && (customInstructions || options?.internalGuidance)) { + throw new Error(`/compact ${compactMode.name} does not take focus instructions.`); + } + const compactionAbortController = new AbortController(); + this.#compactionAbortController = compactionAbortController; + + try { + this.#host.disconnectFromAgent(); + await this.#host.abort({ goalReason: "internal", preserveCompaction: true }); + if (!this.#model) { + throw new Error("No model selected"); + } + + const compactionSettings = this.#host.settings.getGroup("compaction"); + // The `/compact ` override (resolved above) replaces the configured + // strategy/remote flags for this one invocation. Merged before + // prepareCompaction so the remote gating (preparation.settings. + // remoteEnabled/endpoint) and the snapcompact decision below both see it. + const effectiveSettings = compactMode + ? { ...compactionSettings, ...compactMode.overrides } + : compactionSettings; + // /compact remote demands provider-native compaction. When no remote + // endpoint is configured (one would override per-model gating in + // compact()), drop fallback candidates that aren't remote-capable so the + // engine never silently runs a local summary on a configured-but-non- + // remote compactionModel. If filtering empties the chain, warn and fall + // back to the full chain so the operation still completes. + const availableModels = this.#host.modelRegistry.getAvailable(); + const requireProviderRemote = Boolean(compactMode?.requiresRemote && !effectiveSettings.remoteEndpoint); + let compactionCandidates = this.#getCompactionModelCandidates( + availableModels, + requireProviderRemote ? shouldUseOpenAiRemoteCompaction : undefined, + ); + if (requireProviderRemote && compactionCandidates.length === 0) { + this.#host.emitNotice( + "warning", + `remote compaction is unavailable for ${this.#model.id} (no remote endpoint configured and no provider-native remote-capable model in the fallback chain) — using a local summary instead`, + "compaction", + ); + compactionCandidates = this.#getCompactionModelCandidates(availableModels); + } + const pathEntries = this.#host.sessionManager.getBranch(); + const preparation = prepareCompaction(pathEntries, effectiveSettings, this.#model); + if (!preparation) { + // Check why we can't compact + const lastEntry = pathEntries[pathEntries.length - 1]; + if (lastEntry?.type === "compaction") { + throw new Error("Already compacted"); + } + throw new Error("Nothing to compact (session too small)"); + } + + let hookCompaction: CompactionResult | undefined; + let fromExtension = false; + let preserveData: Record | undefined; + + if (this.#host.extensionRunner?.hasHandlers("session_before_compact")) { + const result = (await this.#host.extensionRunner.emit({ + type: "session_before_compact", + preparation, + branchEntries: pathEntries, + customInstructions, + signal: compactionAbortController.signal, + })) as SessionBeforeCompactResult | undefined; + + if (result?.cancel) { + throw new CompactionCancelledError(); + } + + if (result?.compaction) { + hookCompaction = result.compaction; + fromExtension = true; + } + } + + const compactionPrep = await this.#prepareCompactionFromHooks(preparation, hookCompaction); + + // Strategy honored on manual /compact too. Custom instructions (public + // user focus OR internal plan-mode guidance) imply a directed LLM + // summary; a text-only model cannot read snapcompact frames. + const wantsSnapcompact = + compactionPrep.kind !== "fromHook" && + effectiveSettings.strategy === "snapcompact" && + !customInstructions && + !options?.internalGuidance; + // `/compact snapcompact` is an explicit no-LLM archive request: honor + // its contract by failing locally rather than silently shipping the + // transcript to a provider. The default-configured snapcompact + // strategy, in contrast, falls back to LLM compaction (mirroring the + // auto-compaction path) so a routine /compact still completes on a + // text-only model (issue #5064). + const explicitSnapcompact = compactMode?.name === "snapcompact"; + let snapcompactReady = wantsSnapcompact; + const snapcompactShapeSetting = this.#host.settings.get("snapcompact.shape"); + let snapcompactShape: snapcompact.Shape | undefined; + // Claude refuses inputs that reproduce its own reasoning as text + // ("reasoning_extraction"), and the snapcompact archive is replayed as + // text into every later request; drop `¶think:` sections for + // Anthropic-dialect targets (issue #6093). + const snapcompactIncludeThinking = preferredDialect(this.#model.id) !== "anthropic"; + if (wantsSnapcompact && !this.#model.input.includes("image")) { + if (explicitSnapcompact) { + this.#host.emitNotice( + "warning", + `snapcompact needs a vision-capable model (${this.#model.id} is text-only)`, + "compaction", + ); + throw new Error(`snapcompact cannot run locally: ${this.#model.id} is text-only.`); + } + this.#host.emitNotice( + "warning", + `snapcompact needs a vision-capable model (${this.#model.id} is text-only); falling back to LLM compaction`, + "compaction", + ); + snapcompactReady = false; + } else if (snapcompactReady) { + const text = snapcompact.serializeConversation( + convertToLlm(preparation.messagesToSummarize.concat(preparation.turnPrefixMessages)), + { includeThinking: snapcompactIncludeThinking }, + ); + const probeText = snapcompact.renderabilityProbeText( + text, + preparation.previousPreserveData, + preparation.previousSummary, + ); + snapcompactShape = snapcompact.resolveShapeForText(probeText, this.#model, snapcompactShapeSetting); + const renderScan = snapcompact.scanRenderability(probeText, { shape: snapcompactShape }); + if (!renderScan.isSafe) { + const percent = (renderScan.unrenderableRatio * 100).toFixed(1); + this.#host.emitNotice( + "warning", + `snapcompact disabled: unsupported characters for selected snapcompact font (${percent}%). No LLM fallback was attempted.`, + "compaction", + ); + throw new Error( + `snapcompact cannot render this conversation locally: unsupported characters for selected snapcompact font (${percent}%).`, + ); + } + } + + let summary: string; + let shortSummary: string | undefined; + let firstKeptEntryId: string; + let tokensBefore: number; + let details: unknown; + let codexCompaction: CodexCompactionContext | undefined; + + // Snapcompact runs locally first. The frame cap is sized from the live + // model window via #computeSnapcompactMaxFrames so the post-render context + // fits without the warning loop (issue #3247). Zero-frame budget now fails + // the snapcompact request locally rather than falling back to an LLM call. + let snapcompactResult: snapcompact.CompactionResult | undefined; + if (snapcompactReady) { + const maxFrames = this.#computeSnapcompactMaxFrames(preparation, effectiveSettings); + if (maxFrames < 1) { + logger.warn("Snapcompact skipped: kept history alone exceeds the context budget", { + model: this.#model?.id, + }); + this.#host.emitNotice( + "warning", + "snapcompact: kept history alone exceeds the context budget. No LLM fallback was attempted.", + "compaction", + ); + throw new Error("snapcompact cannot run locally: kept history alone exceeds the context budget."); + } else { + const shape = snapcompactShape; + if (!shape) { + throw new Error("snapcompact shape was not resolved before rendering."); + } + snapcompactResult = await snapcompact.compact(preparation, { + convertToLlm, + model: this.#model, + ...(snapcompactShapeSetting === "auto" ? {} : { shape }), + maxFrames, + includeThinking: snapcompactIncludeThinking, + }); + const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); + if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { + logger.warn("Snapcompact exceeded the per-request frame payload budget", { + model: this.#model?.id, + framePayloadBytes, + budget: snapcompact.FRAME_DATA_BYTES_BUDGET, + }); + this.#host.emitNotice( + "warning", + "snapcompact produced too much standing image payload. No LLM fallback was attempted.", + "compaction", + ); + throw new Error( + "snapcompact cannot run locally: standing image payload exceeds the per-request budget.", + ); + } + const ctxWindow = this.#model?.contextWindow ?? 0; + const budget = + ctxWindow > 0 + ? ctxWindow - effectiveReserveTokens(ctxWindow, effectiveSettings) + : Number.POSITIVE_INFINITY; + if (this.#projectSnapcompactContextTokens(preparation, snapcompactResult) > budget) { + logger.warn("Snapcompact still overflows the window after frame-budget sizing", { + model: this.#model?.id, + }); + this.#host.emitNotice( + "warning", + "snapcompact could not bring the context under the limit. No LLM fallback was attempted.", + "compaction", + ); + throw new Error("snapcompact could not bring the context under the limit locally."); + } + } + } + + if (compactionPrep.kind === "fromHook") { + summary = compactionPrep.summary; + shortSummary = compactionPrep.shortSummary; + firstKeptEntryId = compactionPrep.firstKeptEntryId; + tokensBefore = compactionPrep.tokensBefore; + details = compactionPrep.details; + preserveData = compactionPrep.preserveData; + } else if (snapcompactResult) { + summary = snapcompactResult.summary; + shortSummary = snapcompactResult.shortSummary; + firstKeptEntryId = snapcompactResult.firstKeptEntryId; + tokensBefore = snapcompactResult.tokensBefore; + details = snapcompactResult.details; + preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) }; + } else { + codexCompaction = createCodexCompactionContext({ + trigger: "manual", + reason: "user_requested", + phase: "standalone_turn", + }); + // Generate compaction result. Only convert known abort-shaped + // rejections (AbortError raised while the abort signal is set, + // or an already-typed sentinel) into `CompactionCancelledError` + // so downstream callers can discriminate cancel from generic + // failure via `instanceof` without inspecting message strings. + // Real compaction bugs (network, server, parsing, etc.) keep + // their original shape — they must not be silently relabeled + // as cancellations even if the signal happens to be aborted + // for an unrelated reason. Assignments live inside the try + // block because every catch path throws — the post-try reads + // of the result-derived locals are reachable only on success. + try { + const result = await this.#compactWithFallbackModel( + preparation, + options?.internalGuidance ?? customInstructions, + compactionAbortController.signal, + { + promptOverride: this.#host.obfuscateTextForProvider(compactionPrep.hookPrompt), + extraContext: compactionPrep.hookContext, + remoteInstructions: this.#host.baseSystemPrompt().join("\n\n"), + convertToLlm: messages => this.#host.convertToLlmForSideRequest(messages), + codexCompaction, + }, + compactionCandidates, + ); + summary = result.summary; + shortSummary = result.shortSummary; + firstKeptEntryId = result.firstKeptEntryId; + tokensBefore = result.tokensBefore; + details = result.details; + preserveData = mergeLlmCompactionPreserveData(compactionPrep.preserveData, result.preserveData); + } catch (err) { + if (err instanceof CompactionCancelledError) { + throw err; + } + if (compactionAbortController.signal.aborted && err instanceof Error && err.name === "AbortError") { + throw new CompactionCancelledError(); + } + throw err; + } + } + + if (compactionAbortController.signal.aborted) { + throw new CompactionCancelledError(); + } + + this.#host.sessionManager.appendCompaction( + summary, + shortSummary, + firstKeptEntryId, + tokensBefore, + details, + fromExtension, + preserveData, + ); + const newEntries = this.#host.sessionManager.getEntries(); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.rebaseAfterCompaction(); + // Compaction discarded the conversation history that carried the approved + // plan reference. Clear the sent-flag so #buildPlanReferenceMessage re-reads + // the plan from disk and re-injects it on the next turn (issue #1246). + this.#host.resetPlanReference(); + this.#host.resetAdvisorRuntimes(); + this.#host.syncTodoPhasesFromBranch(); + if (codexCompaction) { + this.#host.resetCodexProviderAfterCompaction(codexCompaction); + } else { + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + } + + // Get the saved compaction entry for the hook + const savedCompactionEntry = newEntries.find(e => e.type === "compaction" && e.summary === summary) as + | CompactionEntry + | undefined; + + if (this.#host.extensionRunner && savedCompactionEntry) { + await this.#host.extensionRunner.emit({ + type: "session_compact", + compactionEntry: savedCompactionEntry, + fromExtension, + }); + } + + const compactionResult: CompactionResult = { + summary, + shortSummary, + firstKeptEntryId, + tokensBefore, + details, + preserveData, + }; + options?.onComplete?.(compactionResult); + return compactionResult; + } catch (error) { + const err = error instanceof Error ? error : new Error(String(error)); + options?.onError?.(err); + throw error; + } finally { + if (this.#compactionAbortController === compactionAbortController) { + this.#compactionAbortController = undefined; + } + this.#host.reconnectToAgent(); + // Compaction disconnected before `await abort()`, so abort's finally drain + + // (and any steer/follow-up that arrived mid-compaction — async IRC, an + // `xd://` mount notice, an SDK/RPC steer) was suppressed while disconnected + // (issue #5800). Unlike `/new`/switchSession, compaction preserves the agent + // queues, so nothing else resumes them: re-drain now that the listener is back + // and `isCompacting` is false, or the queued turn hangs until the next prompt. + this.#host.drainStrandedQueuedMessages(); + } + } + + /** + * Ask the active memory backend for an extra-context block to splice into + * the compaction summary prompt. Both the manual and auto compaction paths + * funnel through this helper so the behaviour stays identical. + * + * Failures are swallowed: a memory backend going sideways MUST NOT block + * compaction (which is itself the recovery path for context overflow). + */ + async #collectMemoryBackendContext(preparation: { + messagesToSummarize: AgentMessage[]; + turnPrefixMessages: AgentMessage[]; + }): Promise { + const backend = await resolveMemoryBackend(this.#host.settings); + if (!backend.preCompactionContext) return undefined; + const messages = preparation.messagesToSummarize.concat(preparation.turnPrefixMessages); + try { + return await backend.preCompactionContext(messages, this.#host.settings, this.#host.memoryBackendSession()); + } catch (err) { + logger.debug("Memory backend preCompactionContext failed", { + backend: backend.id, + error: String(err), + }); + return undefined; + } + } + + /** + * Cancel in-progress context maintenance (manual compaction, auto-compaction, or auto-handoff). + */ + abortCompaction(): void { + this.#compactionAbortController?.abort(); + this.#autoCompactionAbortController?.abort(); + this.#host.abortHandoff(); + } + + /** Cancel only automatic maintenance while preserving a manual compaction. */ + abortAutomaticCompaction(): void { + this.#autoCompactionAbortController?.abort(); + } + + /** Trigger idle compaction through the auto-compaction flow (with UI events). */ + async runIdleCompaction(): Promise { + if (this.#host.isStreaming() || this.isCompacting) return; + await this.runAutoCompaction("idle", false, true); + } + + /** + * Local token estimate of the stored conversation (plus any pending messages), + * independent of provider-reported usage. A `before_provider_request` hook + * (e.g. a compression extension such as Headroom) or other on-wire payload + * transform can shrink the request below the real stored conversation; the + * provider then reports deflated prompt tokens, so anchoring the compaction + * decision purely on that usage lets the real history grow unbounded until it + * overflows and native compaction can no longer run. This estimate is the + * floor the compaction decision respects so on-wire compression can never + * suppress it. + */ + #estimateStoredContextTokens(pendingMessages: AgentMessage[] = []): number { + // Exclude encrypted reasoning (thinkingSignature / redactedThinking): its + // local byte size diverges from what the provider bills, so counting it here + // would let a thinking-heavy turn falsely trip the floor. The provider usage + // (the other arm of compactionContextTokens) already accounts for it. + const opts = { excludeEncryptedReasoning: true } as const; + return ( + computeNonMessageTokens(this.#host.nonMessageTokenSource()) + + this.#host.messages().reduce((sum, msg) => sum + estimateTokens(msg, opts), 0) + + pendingMessages.reduce((sum, msg) => sum + estimateTokens(msg, opts), 0) + ); + } + + #estimatePrePromptContextTokens(messages: AgentMessage[], contextWindow: number): number { + const breakdown = this.#host.getContextBreakdown({ contextWindow, pendingMessages: messages }); + const localEstimate = this.#estimateStoredContextTokens(messages); + // Floor by the local estimate: a payload-shrinking before_provider_request + // hook deflates the provider-anchored breakdown, which must not suppress + // pre-prompt compaction (see #estimateStoredContextTokens). + return compactionContextTokens(breakdown?.usedTokens ?? 0, localEstimate); + } + + async runPrePromptCompactionIfNeeded(messages: AgentMessage[]): Promise { + const model = this.#model; + if (!model) return; + const contextWindow = model.contextWindow ?? 0; + if (contextWindow <= 0) return; + const compactionSettings = this.#host.settings.getGroup("compaction"); + const contextTokens = this.#estimatePrePromptContextTokens(messages, contextWindow); + if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) return; + + // Auto-promote first: switching to a larger-context model avoids compacting + // the history at all. The post-turn threshold path already promotes before + // compacting; without this, the pre-prompt path would pre-empt promotion and + // compact (snapcompact/summary) a session that should have just been promoted. + if (await this.#promoteContextModel()) { + logger.debug("Pre-prompt context promotion avoided compaction", { + contextTokens, + contextWindow, + model: `${model.provider}/${model.id}`, + }); + return; + } + + logger.debug("Pre-prompt context maintenance triggered by pending prompt size", { + contextTokens, + contextWindow, + model: `${model.provider}/${model.id}`, + }); + await this.runAutoCompaction("threshold", false, false, false, { + autoContinue: false, + triggerContextTokens: contextTokens, + phase: "pre_turn", + }); + } + + /** + * Compact continuing tool-loop runs before the next provider request. + * + * `onTurnEnd` is the safe boundary: tool results for the just-finished turn + * are already paired in `activeMessages`, the live array the agent loop reads + * before its next model call. Before compacting, the just-finished turn is + * synchronously persisted if async message hooks have not reached the normal + * append path yet. Mid-run handoff is suppressed because resetting the session + * while the loop owns `activeMessages` would race the next request; handoff + * strategy falls back to in-place context-full compaction here. + */ + async maintainContextMidRun( + activeMessages: AgentMessage[], + signal: AbortSignal | undefined, + context: AgentTurnEndContext | undefined, + ): Promise { + if ( + signal?.aborted || + this.#host.isDisposed() || + this.isCompacting || + this.#host.isGeneratingHandoff() || + !context?.willContinue + ) + return; + + const model = this.#model; + const contextWindow = model?.contextWindow ?? 0; + if (contextWindow <= 0) return; + + const compactionSettings = this.#host.settings.getGroup("compaction"); + if ( + !compactionSettings.enabled || + compactionSettings.strategy === "off" || + compactionSettings.midTurnEnabled === false + ) { + return; + } + + const lastAssistant = [...activeMessages] + .reverse() + .find((message): message is AssistantMessage => message.role === "assistant"); + if (!lastAssistant || lastAssistant.stopReason === "aborted" || lastAssistant.stopReason === "error") return; + + if (!(await this.#host.persistTurnMessagesForMidRunCompaction(context))) return; + + const billedContextTokens = calculateContextTokens(lastAssistant.usage); + const storedContextTokens = this.#estimateStoredContextTokens(); + const contextTokens = compactionContextTokens(billedContextTokens, storedContextTokens); + if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) return; + + // Promote to a larger-context sibling before compacting, mirroring the + // pre-prompt (runPrePromptCompactionIfNeeded) and post-turn threshold + // (checkCompaction) paths. Without this, a long mid-turn tool loop that + // crosses the threshold compacts the history (and can hit the no-progress + // dead-end on a single oversized turn) on a model that should have just + // been promoted to a larger window instead. + if (await this.#promoteContextModel()) { + logger.debug("Mid-run context promotion avoided compaction", { + contextTokens, + contextWindow, + from: `${model?.provider}/${model?.id}`, + }); + return; + } + + const messagesBefore = activeMessages.length; + await this.runAutoCompaction("threshold", false, false, false, { + autoContinue: false, + suppressContinuation: true, + suppressHandoff: true, + triggerContextTokens: contextTokens, + phase: "mid_turn", + }); + + if (signal?.aborted) return; + const compactedMessages = this.#host.agent.state.messages; + if (compactedMessages !== activeMessages) { + activeMessages.splice(0, activeMessages.length, ...compactedMessages); + } + logger.debug("Mid-run compaction ran between provider calls", { + contextTokens, + contextWindow, + strategy: compactionSettings.strategy, + goalActive: this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active", + messagesBefore, + messagesAfter: activeMessages.length, + }); + } + /** + * Check if context maintenance or promotion is needed and run it. + * Called after agent_end and before prompt submission. + * + * Four cases (in order): + * 1. Input overflow + promotion: promote to larger model, retry without maintenance. + * 2. Input overflow + no promotion target: run context maintenance, auto-retry on same model. + * 3. Output incomplete (stopReason === "length", e.g. `response.incomplete`): the + * model burned its output budget without producing an actionable deliverable + * (reasoning-only or truncated). Drop the dead turn, try promotion, otherwise + * run compaction/handoff and retry. + * 4. Threshold: context over threshold, run context maintenance (no auto-retry). + * + * @param assistantMessage The assistant message to check + * @param skipAbortedCheck If false, include aborted messages (for pre-prompt check). Default: true + * @param allowDefer If true, threshold-driven handoff strategy may schedule itself as a + * deferred post-prompt task instead of running inline. Callers running inside the + * `agent_end` handler set this to true so `session.prompt()` resolves cleanly; callers + * on the pre-prompt path (where the next agent turn is about to start) set it to false + * to avoid racing the deferred handoff against the new turn. + * @param autoContinue Whether maintenance may schedule the agent-authored continuation prompt. + * @returns whether compaction/recovery scheduled a handoff, retry, auto-continue, or + * queued-message drain that already owns the next turn. Callers MUST skip + * `session_stop` and other agent continuations when `continuationScheduled` + * is true. + */ + async checkCompaction( + assistantMessage: AssistantMessage, + skipAbortedCheck = true, + allowDefer = true, + autoContinue = true, + ): Promise { + // Skip if message was aborted (user cancelled) - unless skipAbortedCheck is false + if (skipAbortedCheck && assistantMessage.stopReason === "aborted") return COMPACTION_CHECK_NONE; + const contextWindow = this.#model?.contextWindow ?? 0; + const generation = this.#host.promptGeneration(); + // Skip overflow check if the message came from a different model. + // This handles the case where user switched from a smaller-context model (e.g. opus) + // to a larger-context model (e.g. codex) - the overflow error from the old model + // shouldn't trigger compaction for the new model. + const sameModel = + this.#model && assistantMessage.provider === this.#model.provider && assistantMessage.model === this.#model.id; + // This handles the case where an error was kept after compaction (in the "kept" region). + // The error shouldn't trigger another compaction since we already compacted. + // Example: opus fails -> switch to codex -> compact -> switch back to opus -> opus error + // is still in context but shouldn't trigger compaction again. + const compactionEntry = getLatestCompactionEntry(this.#host.sessionManager.getBranch()); + const errorIsFromBeforeCompaction = + compactionEntry !== null && assistantMessage.timestamp < new Date(compactionEntry.timestamp).getTime(); + if (sameModel && !errorIsFromBeforeCompaction && AIError.isContextOverflow(assistantMessage, contextWindow)) { + // Clear the failed turn from active context so the retry (or the next + // user prompt) does not replay it. The persisted branch entry stays + // for now: when no recovery path runs, the user-facing transcript + // MUST keep the only assistant message explaining why the turn + // stopped. The branch entry is dropped further down, but only on the + // paths that actually schedule a retry/compaction. + this.#host.removeAssistantMessageFromActiveContext(assistantMessage); + + // Try context promotion first - switch to a larger model and retry without compacting + const promoted = await this.#tryContextPromotion(assistantMessage); + if (promoted) { + await this.#host.dropPersistedAssistantTurn(assistantMessage); + // Retry on the promoted (larger) model without compacting + this.#host.scheduleAgentContinue({ delayMs: 100, generation }); + return COMPACTION_CHECK_CONTINUATION; + } + + // No promotion target available fall through to compaction + const compactionSettings = this.#host.settings.getGroup("compaction"); + if (compactionSettings.enabled && compactionSettings.strategy !== "off") { + return await this.#host.runRecoveryCompactionWithRollback("overflow", assistantMessage, allowDefer, { + autoContinue, + }); + } + return COMPACTION_CHECK_NONE; + } + // A context promotion can land while the failing call is already in + // flight (or on a run whose loop predates the switch): the overflow + // error then arrives stamped with the pre-promotion model while + // `this.#host.model()` is already the promoted target. The sameModel guard + // above deliberately ignores stale foreign-model errors, but this + // state is not stale — recover exactly like the promotion path: + // drop the dead turn and retry on the already-promoted model. Gated + // narrowly on "current model IS the failed model's promotion target + // with a strictly larger window" so genuinely stale errors from + // old user-switched models keep surfacing untouched. + if ( + !sameModel && + autoContinue && + !errorIsFromBeforeCompaction && + assistantMessage.stopReason === "error" && + this.#model && + contextWindow > 0 && + this.#host.settings.getGroup("contextPromotion").enabled + ) { + const failedModel = this.#host.modelRegistry.find(assistantMessage.provider, assistantMessage.model); + const failedWindow = failedModel?.contextWindow ?? 0; + const promotionTarget = failedModel + ? resolveContextPromotionConfiguredTarget(failedModel, this.#host.modelRegistry.getAvailable()) + : undefined; + if ( + failedModel && + failedWindow > 0 && + contextWindow > failedWindow && + promotionTarget && + modelsAreEqual(promotionTarget, this.#model) && + AIError.isContextOverflow(assistantMessage, failedWindow) + ) { + this.#host.removeAssistantMessageFromActiveContext(assistantMessage); + await this.#host.dropPersistedAssistantTurn(assistantMessage); + logger.debug("Overflow on pre-promotion model; retrying on promoted model", { + failed: `${assistantMessage.provider}/${assistantMessage.model}`, + current: `${this.#model.provider}/${this.#model.id}`, + }); + this.#host.scheduleAgentContinue({ delayMs: 100, generation }); + return COMPACTION_CHECK_CONTINUATION; + } + } + + // Case 3: Output-side incomplete — `response.incomplete` from OpenAI Responses + // (and Codex) maps to stopReason === "length". The model burned its + // `max_output_tokens` budget on reasoning/text and emitted no actionable + // deliverable. Same recovery class as overflow: promotion if available, + // otherwise compaction/handoff. Unlike overflow, the *input* is fine, so we + // allow the handoff strategy to actually run. + if (sameModel && !errorIsFromBeforeCompaction && assistantMessage.stopReason === "length") { + // Same active-context vs persisted-history split as the overflow path + // above: clear the dead turn from agent state so it cannot be replayed, + // but keep it on the branch unless promotion or compaction actually runs. + this.#host.removeAssistantMessageFromActiveContext(assistantMessage); + + const promoted = await this.#tryContextPromotion(assistantMessage); + if (promoted) { + await this.#host.dropPersistedAssistantTurn(assistantMessage); + logger.debug("Context promotion triggered by response.incomplete (length stop)", { + from: `${assistantMessage.provider}/${assistantMessage.model}`, + }); + this.#host.scheduleAgentContinue({ delayMs: 100, generation }); + return COMPACTION_CHECK_CONTINUATION; + } + + const incompleteCompactionSettings = this.#host.settings.getGroup("compaction"); + if (incompleteCompactionSettings.enabled && incompleteCompactionSettings.strategy !== "off") { + logger.debug("Compaction triggered by response.incomplete (length stop, no promotion target)", { + model: `${assistantMessage.provider}/${assistantMessage.model}`, + strategy: incompleteCompactionSettings.strategy, + }); + return await this.#host.runRecoveryCompactionWithRollback("incomplete", assistantMessage, allowDefer, { + autoContinue, + triggerContextTokens: calculateContextTokens(assistantMessage.usage), + }); + } + // Neither promotion nor compaction is available — surface the dead-end so + // the user understands why the turn yielded with nothing. + logger.warn("response.incomplete with no recovery path (promotion + compaction both unavailable)", { + model: `${assistantMessage.provider}/${assistantMessage.model}`, + }); + return COMPACTION_CHECK_NONE; + } + + // Stale-result pass runs every turn, before any threshold gating: it is + // cheap (bails when no candidate) and independent of the compaction + // setting. + const supersedeResult = await this.#pruneStaleToolResults(); + + const compactionSettings = this.#host.settings.getGroup("compaction"); + if (!compactionSettings.enabled || compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; + + // Case 4: Threshold - turn succeeded but context is getting large + // Skip if this was an error (non-overflow errors don't have usage data) + if (assistantMessage.stopReason === "error") return COMPACTION_CHECK_NONE; + const pruneResult = await this.#pruneToolOutputs(); + const maintenanceTokensFreed = (supersedeResult?.tokensSaved ?? 0) + (pruneResult?.tokensSaved ?? 0); + // `errorIsFromBeforeCompaction` (computed above) is the general + // "this assistant message predates the latest compaction" predicate here, + // not just an error-specific one; alias it locally so the threshold intent + // reads clearly (#3412 review). + const assistantPredatesCompaction = errorIsFromBeforeCompaction; + // An assistant that predates the latest compaction carries stale, pre-rewrite + // `usage`: the scheduled auto-continue re-enters this check with the kept + // assistant (#promptWithMessage → checkCompaction), and its old high prompt + // count would re-trip the threshold on a freshly compacted history. Drop the + // stale provider number for those messages and let the live stored estimate + // (the floor applied below) drive the decision instead. + const assistantUsageContextTokens = assistantPredatesCompaction + ? 0 + : calculateContextTokens(assistantMessage.usage); + const storedContextTokens = this.#estimateStoredContextTokens(); + // Pruning frees bytes for the NEXT prompt; it does not change the size of + // the prompt the LLM just billed for. Earlier revisions subtracted the + // per-turn supersede/prune `tokensSaved` from the threshold input, which + // let a long-running `/goal` session sit above `compaction.thresholdTokens` + // indefinitely whenever per-turn pruning saved enough to drop the + // post-prune estimate below the user-configured trigger — the visible + // context (anchored to the same provider billing) still showed >threshold, + // but `shouldCompact` no-op'd (#3174). Anchor the initial trigger on the + // last turn's billed context tokens, floored by the post-prune + // stored-conversation estimate so a payload-compression hook still can't + // deflate the trigger. + const contextTokens = compactionContextTokens(assistantUsageContextTokens, storedContextTokens); + const postMaintenanceContextTokens = compactionContextTokens( + Math.max(0, assistantUsageContextTokens - maintenanceTokensFreed), + storedContextTokens, + ); + const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); + const shouldThresholdCompact = shouldCompact(contextTokens, contextWindow, compactionSettings); + logger.debug("Auto-compaction threshold decision", { + phase: "post-agent-end", + goalModeEnabled: this.#goalModeState?.enabled === true, + goalStatus: this.#goalModeState?.goal.status, + stopReason: assistantMessage.stopReason, + sameModel: sameModel === true, + contextWindow, + strategy: compactionSettings.strategy, + thresholdTokens, + assistantUsageContextTokens, + storedContextTokens, + resolvedContextTokens: contextTokens, + postMaintenanceContextTokens, + maintenanceTokensFreed, + shouldCompact: shouldThresholdCompact, + contextPromotionEnabled: this.#host.settings.get("contextPromotion.enabled") === true, + }); + if (shouldThresholdCompact) { + // Try promotion first — if a larger model is available, switch instead of compacting + const promoted = await this.#tryContextPromotion(assistantMessage); + if (!promoted) { + return await this.runAutoCompaction("threshold", false, false, allowDefer, { + autoContinue, + triggerContextTokens: postMaintenanceContextTokens, + phase: "pre_turn", + terminalTextAnswer: isTerminalTextAssistantAnswer(assistantMessage), + }); + } + logger.debug("Auto-compaction threshold satisfied but context promotion took over", { + contextTokens, + contextWindow, + model: `${assistantMessage.provider}/${assistantMessage.model}`, + }); + } + return COMPACTION_CHECK_NONE; + } + + /** + * Attempt context promotion to a larger model. + * Returns true if promotion succeeded (caller should retry without compacting). + */ + async #tryContextPromotion(assistantMessage: AssistantMessage): Promise { + const currentModel = this.#model; + if (!currentModel) return false; + // The overflow/length error may have come from a model the user already + // switched away from; only promote when the failing turn was this model. + if (assistantMessage.provider !== currentModel.provider || assistantMessage.model !== currentModel.id) + return false; + return this.#promoteContextModel(); + } + + /** + * Switch to a larger-context sibling when context promotion is enabled and a + * target with a strictly larger window (and a usable key) exists. Returns true + * when the model was switched, so the caller can retry without compacting. + * Message-independent core shared by the post-turn overflow path + * ({@link #tryContextPromotion}) and the pre-prompt threshold path + * ({@link runPrePromptCompactionIfNeeded}). + */ + async #promoteContextModel(): Promise { + const promotionSettings = this.#host.settings.getGroup("contextPromotion"); + if (!promotionSettings.enabled) return false; + const currentModel = this.#model; + if (!currentModel) return false; + const contextWindow = currentModel.contextWindow ?? 0; + if (contextWindow <= 0) return false; + const targetModel = await this.resolveContextPromotionTarget(currentModel, contextWindow); + if (!targetModel) return false; + + try { + await this.#host.setModelTemporary(targetModel, undefined, { ephemeral: true }); + logger.debug("Context promotion switched model on overflow", { + from: `${currentModel.provider}/${currentModel.id}`, + to: `${targetModel.provider}/${targetModel.id}`, + }); + return true; + } catch (error) { + logger.warn("Context promotion failed", { + from: `${currentModel.provider}/${currentModel.id}`, + to: `${targetModel.provider}/${targetModel.id}`, + error: String(error), + }); + return false; + } + } + + async resolveContextPromotionTarget(currentModel: Model, contextWindow: number): Promise { + const availableModels = this.#host.modelRegistry.getAvailable(); + if (availableModels.length === 0) return undefined; + + const candidate = resolveContextPromotionConfiguredTarget(currentModel, availableModels); + if (!candidate) return undefined; + if (modelsAreEqual(candidate, currentModel)) return undefined; + if (candidate.contextWindow == null || candidate.contextWindow <= contextWindow) return undefined; + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId()); + if (!apiKey) return undefined; + return candidate; + } + + #getCompactionModelCandidates(availableModels: Model[], filter?: (model: Model) => boolean): Model[] { + return this.resolveCompactionModelCandidates(this.#model, availableModels, filter); + } + + resolveCompactionModelCandidates( + preferredModel: Model | null | undefined, + availableModels: Model[], + filter?: (model: Model) => boolean, + ): Model[] { + const candidates: Model[] = []; + const seen = new Set(); + + const addCandidate = (model: Model | undefined): void => { + if (!model) return; + const key = `${model.provider}/${model.id}`; + if (seen.has(key)) return; + seen.add(key); + // `seen` still tracks rejected models so the largest-context fallback + // scan below doesn't reintroduce them; the filter just suppresses + // inclusion in this caller's candidate chain. + if (filter && !filter(model)) return; + candidates.push(model); + }; + + if (preferredModel) { + addCandidate(resolveCompactionConfiguredTarget(preferredModel, availableModels)); + } + addCandidate(preferredModel ?? undefined); + for (const role of MODEL_ROLE_IDS) { + addCandidate( + resolveRoleModelFull(this.#host.settings, role, availableModels, preferredModel ?? undefined).model, + ); + } + + const sortedByContext = [...availableModels].sort((a, b) => (b.contextWindow ?? 0) - (a.contextWindow ?? 0)); + for (const model of sortedByContext) { + if (!seen.has(`${model.provider}/${model.id}`)) { + addCandidate(model); + break; + } + } + + return candidates; + } + + #buildCompactionAuthError(): Error { + const currentModel = this.#model; + if (!currentModel) { + return new Error( + "Compaction requires a model with usable credentials, but no authenticated compaction model is available.", + ); + } + return new Error( + `Compaction requires usable credentials for ${currentModel.provider}/${currentModel.id}. ` + + `Configure ${currentModel.provider} credentials or assign an authenticated fallback role such as modelRoles.smol.`, + ); + } + + async #compactWithFallbackModel( + preparation: CompactionPreparation, + customInstructions: string | undefined, + signal: AbortSignal, + options?: SummaryOptions, + precomputedCandidates?: Model[], + ): Promise { + const candidates = + precomputedCandidates ?? this.#getCompactionModelCandidates(this.#host.modelRegistry.getAvailable()); + const telemetry = resolveTelemetry(this.#host.agent.telemetry, this.#host.sessionId()); + + for (const candidate of candidates) { + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId()); + if (!apiKey) continue; + + try { + return await compact( + this.#host.obfuscatePreparationForProvider(preparation), + candidate, + this.#host.modelRegistry.resolver(candidate, this.#host.sessionId()), + this.#host.obfuscateTextForProvider(customInstructions), + signal, + { + ...options, + metadata: this.#host.agent.metadataForProvider(candidate.provider), + convertToLlm: messages => this.#host.convertToLlmForSideRequest(messages), + telemetry, + // Honor the user's /model thinking selection (incl. `off`) on + // the manual `/compact` path. Clamped per-model inside compact() + // via resolveCompactionEffort so unsupported-effort models + // (xai-oauth/grok-build) don't trip requireSupportedEffort. + thinkingLevel: this.#host.thinkingLevel(), + tools: this.#host.agent.state.tools, + sessionId: this.#host.sessionId(), + promptCacheKey: this.#host.sessionId(), + providerSessionState: this.#host.providerSessionState, + // Route every summarization HTTP request through the + // session's side-stream transport so the provider + // concurrency cap (e.g. providers.ollama-cloud.maxConcurrency) + // brackets compaction the same way it brackets the live + // agent turn — without this, multiple ollama-cloud + // subagents auto/manually compacting issued uncapped + // summary requests in parallel (chatgpt-codex review on + // #3751). + completeImpl: async (requestModel, requestContext, requestOptions) => { + const stream = await this.#host.sideStreamFn(requestModel, requestContext, requestOptions); + return stream.result(); + }, + }, + ); + } catch (error) { + if (!AIError.is(AIError.classify(error, candidate.api), AIError.Flag.AuthFailed)) { + throw error; + } + } + } + + throw this.#buildCompactionAuthError(); + } + + async #prepareCompactionFromHooks( + preparation: CompactionPreparation, + hookCompaction: CompactionResult | undefined, + ): Promise< + | { + kind: "fromHook"; + summary: string; + shortSummary: string | undefined; + firstKeptEntryId: string; + tokensBefore: number; + details: unknown; + preserveData: Record | undefined; + } + | { + kind: "needsLlm"; + hookContext: string[] | undefined; + hookPrompt: string | undefined; + preserveData: Record | undefined; + } + > { + let hookContext: string[] | undefined; + let hookPrompt: string | undefined; + let preserveData: Record | undefined; + + if (!hookCompaction && this.#host.extensionRunner?.hasHandlers("session.compacting")) { + const compactMessages = preparation.messagesToSummarize.concat(preparation.turnPrefixMessages); + const result = (await this.#host.extensionRunner.emit({ + type: "session.compacting", + sessionId: this.#host.sessionId(), + messages: compactMessages, + })) as { context?: string[]; prompt?: string; preserveData?: Record } | undefined; + + hookContext = result?.context; + hookPrompt = result?.prompt; + preserveData = result?.preserveData; + } + + const memoryBackendContext = await this.#collectMemoryBackendContext(preparation); + if (memoryBackendContext) { + hookContext = hookContext ? [...hookContext, memoryBackendContext] : [memoryBackendContext]; + } + + if (hookCompaction) { + preserveData ??= hookCompaction.preserveData; + return { + kind: "fromHook", + summary: hookCompaction.summary, + shortSummary: hookCompaction.shortSummary, + firstKeptEntryId: hookCompaction.firstKeptEntryId, + tokensBefore: hookCompaction.tokensBefore, + details: hookCompaction.details, + preserveData, + }; + } + + return { kind: "needsLlm", hookContext, hookPrompt, preserveData }; + } + + /** + * Cap on snapcompact frames the post-compaction context can carry without + * busting the model window. Mirrors the per-frame token charge used by the + * projection ({@link snapcompact.FRAME_TOKEN_ESTIMATE}, the conservative + * high-res Anthropic ceiling), so picking `maxFrames` from this helper makes + * {@link #projectSnapcompactContextTokens} succeed by construction. + * + * Skip vs. cap use different reserves on purpose. The **skip** decision + * (return `0`) trips only when kept-recent plus non-message tokens already + * eat the entire `ctxWindow − reserve` envelope: at that point no archive + * shape — frame-bearing or text-only — can fit, and the caller MUST + * shortcut to the LLM summarizer instead of re-running snapcompact to + * re-emit the "could not bring the context under the limit" warning every + * threshold tick. The **cap** calculation subtracts a shape-aware reserve + * (`2 × geometry(shape).capacity` chars worth of text edges, billed at the + * tiktoken cl100k baseline, plus a 2k summary-template allowance) sized + * from the same `shape` snapcompact will use, so the projection still + * passes once frames land — but it MUST NOT gate the skip decision, since + * a frame-less archive (`text.length <= 2 * edgeCap` short-circuit in + * `planArchive`) typically costs only a few hundred tokens of summary + * lead and would fit under residual headroom far smaller than the cap + * reserve (chatgpt-codex reviews on #3249). + * + * Returns `1` when the frame charge would overflow but the text-only path + * still has room: snapcompact's planner picks the frame-less layout + * automatically when the discarded text fits in the edges, so giving it + * the minimum cap lets it succeed instead of being skipped outright. + * + * Without this cap, the bundled `MAX_FRAMES_DEFAULT = 80` × 5024 tokens = + * ~402k frame-token projection always overflows any sub-1M-token window + * (issue #3247). + */ + #computeSnapcompactMaxFrames(preparation: CompactionPreparation, settings: CompactionSettings): number { + const ctxWindow = this.#model?.contextWindow ?? 0; + if (ctxWindow <= 0) return Math.min(snapcompact.MAX_FRAMES_DEFAULT, snapcompact.maxFramesForDataBudget()); + const reserve = effectiveReserveTokens(ctxWindow, settings); + let baseTokens = computeNonMessageTokens(this.#host.nonMessageTokenSource()); + for (const message of preparation.recentMessages) { + baseTokens += estimateTokens(message); + } + const totalBudget = ctxWindow - reserve; + // Skip iff there is no headroom whatsoever; a text-only archive costs + // far less than the cap reserve below, so any positive residual is + // worth attempting and the projection guard catches actual overflow. + if (baseTokens >= totalBudget) return 0; + // Cap reserve mirrors what `estimateTokens(summaryMessage)` will charge + // when frames > 0: `countTokens(summaryTemplate ‖ textHead ‖ textTail)` + // plus `numFrames × FRAME_TOKEN_ESTIMATE`. Resolve the shape this + // snapcompact pass will actually use (matches the `shape` argument + // passed to `snapcompact.compact` in the auto and manual paths) so the + // text-edge cost reflects the live frame geometry rather than a fixed + // approximation. Reviewer (chatgpt-codex on #3249): a 4k reserve + // undersized the ~7k text-edge cost on the default Anthropic + // 11on16-bw shape, so the projection then rejected the `maxFrames` + // the cap had picked and the warning loop reappeared. + // + // - `textHead` and `textTail` each consume up to `geometry.capacity` + // chars when frames > 0 (one HQ-capacity page per edge: see + // `TEXT_EDGE_PAGES = 1` in `planArchive`), so 2 × capacity chars + // total. Per-shape capacity: Anthropic 11on16-bw ~13.9k, Opus + // 1932px ~21k, Gemini 8on22-bw 2048px ~23.8k, OpenAI 1568px ~13.9k. + // - tiktoken cl100k ≈ 4 chars/token on ASCII (verified empirically + // for prose, code, and JSON); a 1.15 multiplier absorbs tokenizer + // drift on denser content (e.g. dense JSON / tool-result blobs). + // - Summary template (intro + FILES section + grid notes) bills + // ~2k tokens for typical sessions. + const shape = snapcompact.resolveShape(this.#model, this.#host.settings.get("snapcompact.shape")); + const edgeCap = snapcompact.geometry(shape).capacity; + const textEdgeTokens = Math.ceil((2 * edgeCap * 1.15) / 4); + const SUMMARY_TEMPLATE_TOKENS = 2000; + const capReserve = textEdgeTokens + SUMMARY_TEMPLATE_TOKENS; + const frameBudget = totalBudget - baseTokens - capReserve; + if (frameBudget < snapcompact.FRAME_TOKEN_ESTIMATE) return 1; + return Math.min( + Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), + snapcompact.MAX_FRAMES_DEFAULT, + snapcompact.maxFramesForDataBudget(), + ); + } + + #snapcompactFramePayloadBytes(result: snapcompact.CompactionResult): number { + const archive = snapcompact.getPreservedArchive(result.preserveData); + return archive ? snapcompact.frameDataBytes(archive.frames) : 0; + } + + /** + * Project the post-compaction context size of a snapcompact result: kept + * recent messages + the summary message with its re-attached frames + the + * fixed non-message overhead (system prompt + tools). Mirrors how the + * compacted context is rebuilt, so the estimate matches the wire shape, and + * lets the caller decide whether snapcompact brought the context under the + * window or should fall back to an LLM summary. + */ + #projectSnapcompactContextTokens(preparation: CompactionPreparation, result: snapcompact.CompactionResult): number { + const archive = snapcompact.getPreservedArchive(result.preserveData); + const blocks = archive + ? snapcompact.historyBlocks(archive, { maxFrameDataBytes: snapcompact.FRAME_DATA_BYTES_BUDGET }) + : undefined; + const summaryMessage = createCompactionSummaryMessage( + result.summary, + result.tokensBefore, + new Date().toISOString(), + result.shortSummary, + undefined, + undefined, + blocks, + ); + let tokens = computeNonMessageTokens(this.#host.nonMessageTokenSource()) + estimateTokens(summaryMessage); + for (const message of preparation.recentMessages) { + tokens += estimateTokens(message); + } + return tokens; + } + + /** + * Post-maintenance progress check for the context-full / snapcompact tail. + * + * After `appendCompaction` rewrote history and `replaceMessages` swapped in the + * compacted context, measure the residual context off the live message set and + * decide whether maintenance actually created headroom. Mirrors the shake + * recovery-band logic (#2275): a session whose single most-recent turn already + * blows the threshold cannot be reduced by compaction (findCutPoint keeps that + * turn verbatim), so re-firing on the next agent_end just thrashes. We only + * report progress when residual context lands at or below + * `COMPACTION_RECOVERY_BAND × threshold` — a band that sits strictly under the + * compaction threshold, so reaching it guarantees the next turn cannot + * re-trip threshold compaction. + * + * When the model/window is unknown we cannot evaluate the band, so we + * optimistically allow the continuation (preserving prior behavior). + */ + #compactionCreatedHeadroom(): boolean { + const contextWindow = this.#model?.contextWindow ?? 0; + if (contextWindow <= 0) return true; + const compactionSettings = this.#host.settings.getGroup("compaction"); + const residualTokens = compactionContextTokens( + this.#host.getContextUsage({ contextWindow })?.tokens ?? 0, + this.#estimateStoredContextTokens(), + ); + const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); + const recoveryBand = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); + // Residual at/below the band is authoritative headroom: the band sits + // strictly under the compaction threshold, so the next turn cannot + // re-trip threshold compaction regardless of how little this pass shaved. + // Don't add a secondary "smaller than the trigger" guard — when stale/ + // tool-output pruning already dropped context under the band before this + // pass, the trigger is itself sub-band, and requiring a strict reduction + // would suppress a valid continuation and emit a false no-progress warning + // even though compaction left the session safe. + return residualTokens <= recoveryBand; + } + + /** + * Retry-side counterpart to {@link #compactionCreatedHeadroom}. An + * overflow/incomplete recovery only needs the rebuilt prompt to *fit* the + * window again — it does not have to land under the compaction threshold, let + * alone the stricter `COMPACTION_RECOVERY_BAND × threshold` hysteresis the + * auto-continue thrash guard uses. Reusing the band here turned recoverable + * overflows into manual dead-ends: a 200k-window prompt compacted from + * overflow down to ~150k is comfortably retryable, but sits above + * `0.8 × 170k = 136k` and was wrongly refused (PR #3412 review). + * + * Measures residual context against the usable budget (`contextWindow - reserve`). + * The default absolute reserve can exceed bundled small-context windows, or + * nearly consume a 16k-class window; those known-impossible defaults fall + * back to the proportional 15% reserve. Explicit valid reserves still define + * the usable prompt budget so retries do not enter headroom the user + * intentionally reserved. Callers MUST + * invoke this AFTER dropping the failed assistant from `this.#host.messages()`, so + * the just-failed turn (which the retry prompt will not include) is excluded + * from the estimate. + * + * When the model/window is unknown we cannot evaluate the budget, so we + * optimistically allow the retry (preserving prior behavior). + */ + #compactionCreatedRetryFit(): boolean { + const contextWindow = this.#model?.contextWindow ?? 0; + if (contextWindow <= 0) return true; + const compactionSettings = this.#host.settings.getGroup("compaction"); + const residualTokens = compactionContextTokens( + this.#host.getContextUsage({ contextWindow })?.tokens ?? 0, + this.#estimateStoredContextTokens(), + ); + const fitBudget = Math.max(0, contextWindow - resolveBudgetReserveTokens(contextWindow, compactionSettings)); + return residualTokens <= fitBudget; + } + + /** + * Last-resort tiered reducer when {@link runAutoCompaction} would otherwise + * dead-end. The summarizer cut at the only available turn boundary, but the + * kept tail is still over the recovery band because a single recent turn (a + * large tool-result, a heavy fenced/XML block, attached images) is itself + * bigger than the band and `findCutPoint` cannot cut inside one message. + * + * Tier 1 — `shake("elide")` reaches INSIDE that tail: heavy tool-result / + * block content is offloaded to one `artifact://` blob behind a recoverable + * placeholder. Skipped when this pass already ran a shake (`skipElide`). + * Tier 2 — `dropImages()`: the manual `/shake images` remedy, automated. + * Image blocks are stripped from the branch; unlike elided text they are NOT + * artifact-recoverable, so this tier only runs once elide has failed the + * progress re-test. + * + * Each tier that rewrote history re-anchors the in-flight context snapshot, + * then the caller's progress predicate is re-tested; the first tier that + * restores progress emits one info notice describing everything freed and + * stops. Returns whether progress was restored — `false` falls through to + * the dead-end warning. + */ + async #rescueCompactionDeadEnd( + signal: AbortSignal, + options: { skipElide: boolean; hasProgress: () => boolean }, + ): Promise { + if (signal.aborted) return false; + // Tier 0 — a snapcompact pass whose just-written frame archive is itself + // the over-budget cost (each pass re-renders the carried-forward text + // into MORE frames, so the archive grows past the recovery band and the + // elide/image tiers below can never shrink it): rebuild the archive at + // a threshold-derived frame budget. + const frameRescue = await this.#rescueSnapcompactFrameOverflow( + this.#host.sessionManager.getBranch(), + this.#host.settings.getGroup("compaction"), + signal, + ); + if (frameRescue !== undefined && options.hasProgress()) return true; + let elided = 0; + let elidedTokens = 0; + let elideSink = "placeholders"; + if (!options.skipElide) { + try { + const result = await this.#host.shake("elide", { signal }); + elided = result.toolResultsDropped + result.blocksDropped; + elidedTokens = result.tokensFreed; + if (result.artifactId) elideSink = "an artifact"; + if (elided > 0) { + // The elide pass rewrote history; re-anchor the in-flight snapshot + // so the caller's headroom/retry-fit re-test measures the shaken + // context. + this.#host.rebaseAfterCompaction(); + } + } catch (error) { + logger.warn("Dead-end shake rescue failed", { + error: error instanceof Error ? error.message : String(error), + }); + } + if (elided > 0 && options.hasProgress()) { + this.#host.emitNotice( + "info", + `Compaction dead-end recovery: ${this.#describeElideRescue(elided, elidedTokens, elideSink)} so maintenance could make progress.`, + "compaction", + ); + return true; + } + } + if (signal.aborted) return false; + let imagesDropped = 0; + try { + imagesDropped = (await this.#host.dropImages()).removed; + if (imagesDropped > 0) this.#host.rebaseAfterCompaction(); + } catch (error) { + logger.warn("Dead-end image-drop rescue failed", { + error: error instanceof Error ? error.message : String(error), + }); + } + if (imagesDropped > 0 && options.hasProgress()) { + const elidedPart = elided > 0 ? `${this.#describeElideRescue(elided, elidedTokens, elideSink)} and ` : ""; + this.#host.emitNotice( + "info", + `Compaction dead-end recovery: ${elidedPart}dropped ${imagesDropped} attached image${imagesDropped === 1 ? "" : "s"} so maintenance could make progress.`, + "compaction", + ); + return true; + } + return false; + } + + /** Notice fragment for a dead-end elide tier: what was freed and where it went. */ + #describeElideRescue(elided: number, tokensFreed: number, sink: string): string { + return `elided ${elided} heavy block${elided === 1 ? "" : "s"} (~${tokensFreed.toLocaleString()} tokens) to ${sink}`; + } + + /** + * Frame budget for {@link #rescueSnapcompactFrameOverflow}: targets + * `COMPACTION_RECOVERY_BAND × threshold` (the same band + * {@link #compactionCreatedHeadroom} re-tests), not the window-fit budget + * {@link #computeSnapcompactMaxFrames} sizes against — a rebuilt archive + * must land back under the maintenance trigger, or the next settle + * re-enters the same dead-end. Cap reserve mirrors + * #computeSnapcompactMaxFrames (text edges + summary template), and + * `keptTailTokens` charges the kept entries AFTER the archive so the + * budget mirrors what #compactionCreatedHeadroom will actually measure. + * Returns 0 when not even one frame fits that budget — the rebuild could + * never create headroom, so the caller must not append it. + */ + #computeSnapcompactRescueMaxFrames(settings: CompactionSettings, keptTailTokens: number): number { + const ctxWindow = this.#model?.contextWindow ?? 0; + if (ctxWindow <= 0) return Math.min(snapcompact.MAX_FRAMES_DEFAULT, snapcompact.maxFramesForDataBudget()); + const thresholdTokens = resolveThresholdTokens(ctxWindow, settings); + const recoveryBandTokens = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); + const baseTokens = computeNonMessageTokens(this.#host.nonMessageTokenSource()); + const shape = snapcompact.resolveShape(this.#model, this.#host.settings.get("snapcompact.shape")); + const edgeCap = snapcompact.geometry(shape).capacity; + const textEdgeTokens = Math.ceil((2 * edgeCap * 1.15) / 4); + const SUMMARY_TEMPLATE_TOKENS = 2000; + const frameBudget = recoveryBandTokens - baseTokens - keptTailTokens - textEdgeTokens - SUMMARY_TEMPLATE_TOKENS; + if (frameBudget < snapcompact.FRAME_TOKEN_ESTIMATE) return 0; + // Same hard caps as #computeSnapcompactMaxFrames: a threshold-derived + // count above the per-request payload budget would "shrink" a huge + // archive to a frame count the rebuilt prompt can never attach anyway. + return Math.min( + Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), + snapcompact.MAX_FRAMES_DEFAULT, + snapcompact.maxFramesForDataBudget(), + ); + } + + /** + * Dead-end rescue for a branch whose latest snapcompact CompactionEntry is + * itself billed past the maintenance threshold + * (`FRAME_TOKEN_ESTIMATE × frames`). Reaching the `!preparation` dead-end + * proves everything after that entry is already kept-recent (nothing to + * summarize), so the archive is the irreducible cost — and the elide/image + * tiers can never touch it: `collectShakeRegions` and `dropImages()` only + * inspect "message"/"custom_message" entries, so a `type: "compaction"` + * entry falls through both and the session re-warns on every resume (the + * shape issue #4786's rescue does not cover). + * + * Rebuilds the SAME archive locally — no LLM, no network — by re-running + * `snapcompact.compact()` over the entry's carried-forward source text at + * a maxFrames derived from the trigger threshold instead of the window: + * `planArchive` truncates the oldest chars to fit, so the rebuilt entry + * genuinely shrinks. The rebuilt entry keeps the stale entry's + * `firstKeptEntryId`, so the kept tail is untouched, and persisting + * through `appendCompaction()` lets the write-time superseded-compaction + * elision drop the stale frame payload from the JSONL automatically. + */ + async #rescueSnapcompactFrameOverflow( + branchEntries: SessionEntry[], + settings: CompactionSettings, + signal: AbortSignal, + ): Promise { + if (signal.aborted) return undefined; + // Re-rendering frames needs a vision-capable model, same gate as the + // snapcompact strategy path. + if (!this.#model?.input.includes("image")) return undefined; + const staleEntry = getLatestCompactionEntry(branchEntries); + if (!staleEntry) return undefined; + // Only rescue when the archive is the actual source of the overflow. + // The frame budget below charges every kept entry the rebuilt context + // will still carry — the kept-recent region from `firstKeptEntryId` + // (re-emitted before the archive by buildSessionContext) plus the + // entries after the archive — on top of the fixed context, mirroring + // what #compactionCreatedHeadroom will measure. When not even one + // frame fits (e.g. a huge kept tool result dominates), rebuilding + // would append the replacement compaction at the leaf — turning the + // branch tail into a compaction entry, which prepareCompaction's + // last-entry guard can never summarize past even after an elide + // shrinks the real culprit. Bail and let the elide/image tiers handle + // that tail instead. + let keptTailTokens = 0; + let inKeptRegion = false; + for (const entry of branchEntries) { + if (entry.id === staleEntry.firstKeptEntryId) inKeptRegion = true; + if (entry.id === staleEntry.id) { + // Everything after the archive is always kept. + inKeptRegion = true; + continue; + } + if (!inKeptRegion) continue; + const message = (entry as { message?: AgentMessage }).message; + if (message) keptTailTokens += estimateTokens(message); + } + const archive = snapcompact.getPreservedArchive(staleEntry.preserveData); + if (!archive || archive.frames.length <= 1) return undefined; + const archiveText = snapcompact.archiveSourceText(archive); + if (!archiveText) return undefined; + const maxFrames = this.#computeSnapcompactRescueMaxFrames(settings, keptTailTokens); + if (maxFrames < 1 || maxFrames >= archive.frames.length) return undefined; + + const staleDetails = staleEntry.details as snapcompact.CompactionDetails | undefined; + const fileOps = snapcompact.createFileOps(); + for (const file of staleDetails?.readFiles ?? []) fileOps.read.add(file); + for (const file of staleDetails?.modifiedFiles ?? []) fileOps.edited.add(file); + const shapeSetting = this.#host.settings.get("snapcompact.shape"); + const shape = snapcompact.resolveShapeForText(archiveText, this.#model, shapeSetting); + let result: snapcompact.CompactionResult; + try { + result = await snapcompact.compact( + { + firstKeptEntryId: staleEntry.firstKeptEntryId, + messagesToSummarize: [], + turnPrefixMessages: [], + tokensBefore: staleEntry.tokensBefore, + previousSummary: staleEntry.summary, + previousPreserveData: staleEntry.preserveData, + fileOps, + }, + { + convertToLlm, + model: this.#model, + ...(shapeSetting === "auto" ? {} : { shape }), + maxFrames, + }, + ); + } catch (error) { + logger.warn("Dead-end snapcompact frame rescue failed", { + error: error instanceof Error ? error.message : String(error), + }); + return undefined; + } + if (signal.aborted) return undefined; + const rebuilt = snapcompact.getPreservedArchive(result.preserveData); + if (!rebuilt || rebuilt.frames.length >= archive.frames.length) return undefined; + + const rebuiltEntryId = this.#host.sessionManager.appendCompaction( + result.summary, + result.shortSummary, + result.firstKeptEntryId, + result.tokensBefore, + result.details, + false, + result.preserveData, + ); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.rebaseAfterCompaction(); + // Same post-rewrite bookkeeping as the regular compaction append: the + // rebuilt context no longer carries the transient plan reference (#1246), + // and advisor cursors / todo phases were derived from the replaced + // history. + this.#host.resetPlanReference(); + this.#host.resetAdvisorRuntimes(); + this.#host.syncTodoPhasesFromBranch(); + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + // Extensions must see the entry that is now active, not (only) the one + // this rebuild just superseded — mirror the regular append path's hook. + const rebuiltEntry = this.#host.sessionManager.getEntries().find(e => e.id === rebuiltEntryId) as + | CompactionEntry + | undefined; + if (this.#host.extensionRunner && rebuiltEntry) { + await this.#host.extensionRunner.emit({ + type: "session_compact", + compactionEntry: rebuiltEntry, + fromExtension: false, + }); + } + this.#host.emitNotice( + "info", + `Compaction dead-end recovery: rebuilt the trailing snapcompact archive at a smaller frame budget (${archive.frames.length} → ${rebuilt.frames.length} frames) so maintenance could make progress.`, + "compaction", + ); + return result; + } + + /** + * Internal: Run auto-compaction with events. + * + * @param allowDefer If true (default), threshold-driven handoff strategy is allowed to + * schedule itself as a deferred post-prompt task and return a deferred-handoff result + * immediately. The caller MUST treat that as "compaction will happen async — do not + * also schedule `agent.continue()` for this turn", otherwise the deferred handoff + * races a fresh streaming turn (the symptom: "Auto-handoff" loader + assistant + * message still streaming). Callers on a path that is about to start a new agent + * turn (e.g. the pre-prompt check in `#promptWithMessage`) pass `false` to force + * inline execution so the handoff completes before the new turn begins. + * @returns whether auto-compaction scheduled a follow-up turn. + */ + async runAutoCompaction( + reason: "overflow" | "threshold" | "idle" | "incomplete", + willRetry: boolean, + deferred = false, + allowDefer = true, + options: { + autoContinue?: boolean; + triggerContextTokens?: number; + suppressContinuation?: boolean; + suppressHandoff?: boolean; + phase?: CodexCompactionContext["phase"]; + terminalTextAnswer?: boolean; + } = {}, + ): Promise { + const compactionSettings = this.#host.settings.getGroup("compaction"); + if (compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; + if (reason !== "idle" && !compactionSettings.enabled) return COMPACTION_CHECK_NONE; + const generation = this.#host.promptGeneration(); + const terminalTextAnswer = + options.terminalTextAnswer ?? isTerminalTextAssistantAnswer(this.#host.findLastAssistantMessage()); + const suppressContinuation = options.suppressContinuation === true; + const shouldAutoContinue = + !suppressContinuation && options.autoContinue !== false && compactionSettings.autoContinue !== false; + const suppressHandoff = options.suppressHandoff === true; + let fallbackFromShake = false; + // Shake runs inline (cheap, no remote LLM). On overflow recovery, if shake + // reclaims nothing we fall through to the summary-compaction body below so + // the oversized input still gets resolved. + if (compactionSettings.strategy === "shake") { + const outcome = await this.#runAutoShake( + reason, + willRetry, + generation, + shouldAutoContinue, + terminalTextAnswer, + options.triggerContextTokens, + suppressContinuation, + ); + if (outcome !== "fallback") return outcome; + fallbackFromShake = true; + } + // "overflow" and "incomplete" force inline execution because they are recovery + // paths the caller wants resolved before scheduling the next turn. "idle" is + // triggered by the idle loop and does its own scheduling. + if ( + !suppressHandoff && + !deferred && + allowDefer && + reason !== "overflow" && + reason !== "incomplete" && + reason !== "idle" && + compactionSettings.strategy === "handoff" + ) { + this.#host.schedulePostPromptTask( + async signal => { + await Promise.resolve(); + if (signal.aborted) return; + await this.runAutoCompaction(reason, willRetry, true, true, { + ...options, + terminalTextAnswer, + }); + }, + { generation }, + ); + return { + ...COMPACTION_CHECK_DEFERRED_HANDOFF, + continuationScheduled: shouldAutoContinue, + }; + } + + // "overflow" forces context-full because the input itself is broken — a handoff + // LLM call would hit the same overflow. "incomplete" is an output-side problem, + // so a handoff request on the existing context is still viable. + let action: "context-full" | "handoff" | "snapcompact" = + compactionSettings.strategy === "snapcompact" + ? "snapcompact" + : compactionSettings.strategy === "handoff" && reason !== "overflow" && !suppressHandoff + ? "handoff" + : "context-full"; + if (action === "snapcompact" && this.#model && !this.#model.input.includes("image")) { + this.#host.emitNotice( + "warning", + `snapcompact needs a vision-capable active model (${this.#model.id} is text-only); using context-full auto-compaction instead.`, + "compaction", + ); + action = "context-full"; + } + // Abort any older auto-compaction before installing this run's controller. + this.#autoCompactionAbortController?.abort(); + const autoCompactionAbortController = new AbortController(); + this.#autoCompactionAbortController = autoCompactionAbortController; + const autoCompactionSignal = autoCompactionAbortController.signal; + + try { + // Emit start AFTER the controller is installed so isCompacting is already true + // for any listener — and for input routed during this emit's event-loop yield: + // a message typed as the compaction loader appears must land in the compaction + // queue, not the core steering queue (which handoff's agent.reset() would wipe). + await this.#host.emitSessionEvent({ type: "auto_compaction_start", reason, action }); + if (action === "handoff") { + let handoffSwitchCancelled = false; + const handoffFocus = AUTO_HANDOFF_THRESHOLD_FOCUS; + const handoffResult = await this.#host.runHandoff(handoffFocus, { + autoTriggered: true, + signal: autoCompactionSignal, + onSwitchCancelled: () => { + handoffSwitchCancelled = true; + }, + }); + if (!handoffResult) { + const aborted = autoCompactionSignal.aborted || handoffSwitchCancelled; + if (aborted) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: true, + willRetry: false, + }); + return COMPACTION_CHECK_NONE; + } + logger.warn("Auto-handoff returned no document; falling back to context-full maintenance", { + reason, + }); + action = "context-full"; + } + if (handoffResult) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + }); + const continuationScheduled = + !autoCompactionSignal.aborted && + this.#host.scheduleCompactionContinuation({ + generation, + autoContinue: reason !== "idle" && shouldAutoContinue, + terminalTextAnswer, + suppressContinuation, + }); + return { + ...(continuationScheduled ? COMPACTION_CHECK_CONTINUATION : COMPACTION_CHECK_NONE), + historyRewritten: true, + }; + } + } + + if (!this.#model) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + skipped: true, + }); + return COMPACTION_CHECK_NONE; + } + + const availableModels = this.#host.modelRegistry.getAvailable(); + if (availableModels.length === 0) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + skipped: true, + }); + return COMPACTION_CHECK_NONE; + } + + const pathEntries = this.#host.sessionManager.getBranch(); + + let pathEntriesForCompaction = pathEntries; + let preparation = prepareCompaction(pathEntriesForCompaction, compactionSettings, this.#model); + if (!preparation) { + // prepareCompaction found nothing to summarize because the kept region + // is a single oversized recent turn — findCutPoint never cuts inside a + // tool result, so a huge tool-result / fenced block tail leaves nothing + // on the summarizable side and summary compaction cannot even start. + // That is exactly the dead-end the elide shake rescues: it reaches + // INSIDE the tail and offloads heavy content to an artifact placeholder, + // shrinking the tail so findCutPoint can then move the cut and leave + // older turns to summarize. Run the same tiered rescue the + // post-maintenance guard uses (elide, then image drop), with progress + // defined as "prepareCompaction now succeeds on the rewritten branch", + // and fall through to the normal compaction body when it does (writing + // a compaction entry anchors the stale billed usage so the + // auto-continue re-check cannot re-trip and loop the warning — issue + // #4786). `skipElide` when we already fell through from a shake + // strategy pass (it tried and found nothing); skip entirely on the + // idle timer (it re-checks usage on its own cadence). + let rescueRewroteHistory = false; + // A snapcompact CompactionEntry is invisible to both rescue tiers + // below (they only inspect message entries) and to prepareCompaction + // itself (last-entry-is-compaction guard), so a frame archive billed + // past the threshold dead-ends here on every resume. Rebuild it at a + // threshold-derived frame budget first — but treat that as complete + // only when it actually created headroom: the latest archive may not + // be the oversized tail (e.g. a huge kept tool result after it), and + // declaring victory on a mere frame-count shrink would skip the + // elide/image tiers that can still reach that tail and suppress a + // warning the user should see. + let frameRescueResult: snapcompact.CompactionResult | undefined; + let frameRescueCreatedHeadroom = false; + if (reason !== "idle") { + frameRescueResult = await this.#rescueSnapcompactFrameOverflow( + pathEntriesForCompaction, + compactionSettings, + autoCompactionSignal, + ); + if (frameRescueResult) { + rescueRewroteHistory = true; + pathEntriesForCompaction = this.#host.sessionManager.getBranch(); + frameRescueCreatedHeadroom = this.#compactionCreatedHeadroom(); + } + if (!frameRescueCreatedHeadroom) { + await this.#rescueCompactionDeadEnd(autoCompactionSignal, { + skipElide: fallbackFromShake, + hasProgress: () => { + // Only reached when a tier actually freed something, so the + // branch has been rewritten either way. + rescueRewroteHistory = true; + pathEntriesForCompaction = this.#host.sessionManager.getBranch(); + preparation = prepareCompaction(pathEntriesForCompaction, compactionSettings, this.#model); + return preparation !== undefined; + }, + }); + } + } + if (!preparation) { + const noProgressDeadEnd = reason !== "idle" && !frameRescueCreatedHeadroom; + const deadEndWarning = noProgressDeadEnd + ? compactionDeadEndWarning("shrink it (e.g. clear large tool output)") + : undefined; + // A rescue that appended a rebuilt archive without creating + // headroom must carry the dead-end badge on the entry the + // transcript actually shows (the rebuilt one), or the pause + // loses its explanation once the notice scrolls away. Stamp it + // BEFORE the auto_compaction_end event: a result-carrying event + // makes the TUI rebuild the chat from the current entries + // immediately, so a later stamp would not appear until some + // unrelated rebuild. + if (deadEndWarning && frameRescueResult) { + const stampEntry = getLatestCompactionEntry(this.#host.sessionManager.getBranch()); + if (stampEntry) { + stampEntry.warning = deadEndWarning; + await this.#host.sessionManager.rewriteEntries(); + } + } + // A successful frame rescue rewrote history and activated a new + // compaction entry — surface it as a real (non-skipped) result so + // the TUI rebuilds the transcript instead of treating the pass as + // a benign no-op. + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: frameRescueResult, + aborted: false, + willRetry: false, + skipped: frameRescueResult === undefined, + }); + let continuationScheduled = false; + if (frameRescueCreatedHeadroom) { + continuationScheduled = this.#host.scheduleCompactionContinuation({ + generation, + autoContinue: shouldAutoContinue, + terminalTextAnswer, + suppressContinuation, + }); + } else if (!suppressContinuation && this.#host.agent.hasQueuedMessages()) { + this.#host.scheduleAgentContinue({ + delayMs: 100, + generation, + shouldContinue: () => this.#host.agent.hasQueuedMessages(), + }); + continuationScheduled = true; + } + if (deadEndWarning) { + this.#host.emitNotice("warning", deadEndWarning, "compaction"); + } + // A rescue that offloaded content but still could not produce a + // preparation rewrote the branch; flag it so the overflow-recovery + // rollback does not re-restore the just-failed assistant turn on top + // of the elided tail. + const base = continuationScheduled + ? COMPACTION_CHECK_CONTINUATION + : noProgressDeadEnd + ? COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION + : COMPACTION_CHECK_NONE; + return rescueRewroteHistory ? { ...base, historyRewritten: true } : base; + } + } + + let hookCompaction: CompactionResult | undefined; + let fromExtension = false; + let preserveData: Record | undefined; + let codexCompaction: CodexCompactionContext | undefined; + + if (this.#host.extensionRunner?.hasHandlers("session_before_compact")) { + const hookResult = (await this.#host.extensionRunner.emit({ + type: "session_before_compact", + preparation, + branchEntries: pathEntriesForCompaction, + customInstructions: undefined, + signal: autoCompactionSignal, + })) as SessionBeforeCompactResult | undefined; + + if (hookResult?.cancel) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: true, + willRetry: false, + }); + return COMPACTION_CHECK_NONE; + } + + if (hookResult?.compaction) { + hookCompaction = hookResult.compaction; + fromExtension = true; + } + } + + const compactionPrep = await this.#prepareCompactionFromHooks(preparation, hookCompaction); + + let summary: string; + let shortSummary: string | undefined; + let firstKeptEntryId: string; + let tokensBefore: number; + let details: unknown; + + // Snapcompact runs locally first. The post-compaction context = kept-recent + // + a summary message carrying the imaged archive at FRAME_TOKEN_ESTIMATE + // per frame; #computeSnapcompactMaxFrames sizes the frame cap from the + // live window so we don't run snapcompact just to overflow every threshold + // tick. Any local blocker (unsupported snapcompact glyphs, kept-history too large, + // post-render overflow) downgrades auto maintenance to a context-full LLM + // summary instead of wedging the session (#3659) — auto runs the default + // strategy on the user's behalf, so a fallback that lets the session keep + // running is the right behavior. Manual `/compact snapcompact` keeps the + // local-only contract (#3599): the user explicitly picked it. + let snapcompactResult: snapcompact.CompactionResult | undefined; + let snapcompactBlocker: string | undefined; + if (action === "snapcompact" && compactionPrep.kind !== "fromHook") { + // Drop `¶think:` sections for Anthropic-dialect targets: the archive + // is replayed as text and Claude refuses reproduced reasoning + // ("reasoning_extraction", issue #6093). + const snapcompactIncludeThinking = preferredDialect(this.#model.id) !== "anthropic"; + const text = snapcompact.serializeConversation( + convertToLlm(preparation.messagesToSummarize.concat(preparation.turnPrefixMessages)), + { includeThinking: snapcompactIncludeThinking }, + ); + const probeText = snapcompact.renderabilityProbeText( + text, + preparation.previousPreserveData, + preparation.previousSummary, + ); + const shapeSetting = this.#host.settings.get("snapcompact.shape"); + const shape = snapcompact.resolveShapeForText(probeText, this.#model, shapeSetting); + const renderScan = snapcompact.scanRenderability(probeText, { shape }); + if (!renderScan.isSafe) { + const percent = (renderScan.unrenderableRatio * 100).toFixed(1); + logger.warn("Snapcompact disabled: unsupported characters for selected snapcompact font", { + model: this.#model?.id, + unrenderableRatio: renderScan.unrenderableRatio, + }); + snapcompactBlocker = `snapcompact disabled: unsupported characters for selected snapcompact font (${percent}%); using context-full auto-compaction instead.`; + } else { + const maxFrames = this.#computeSnapcompactMaxFrames(preparation, compactionSettings); + if (maxFrames < 1) { + logger.warn("Snapcompact skipped: kept history alone exceeds the context budget", { + model: this.#model?.id, + }); + snapcompactBlocker = + "snapcompact: kept history alone exceeds the context budget; using context-full auto-compaction instead."; + } else { + snapcompactResult = await snapcompact.compact(preparation, { + convertToLlm, + model: this.#model, + ...(shapeSetting === "auto" ? {} : { shape }), + maxFrames, + includeThinking: snapcompactIncludeThinking, + }); + const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); + if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { + logger.warn("Snapcompact exceeded the per-request frame payload budget", { + model: this.#model?.id, + framePayloadBytes, + budget: snapcompact.FRAME_DATA_BYTES_BUDGET, + }); + snapcompactBlocker = + "snapcompact produced too much standing image payload; using context-full auto-compaction instead."; + snapcompactResult = undefined; + } + if (snapcompactResult) { + const ctxWindow = this.#model?.contextWindow ?? 0; + const budget = + ctxWindow > 0 + ? ctxWindow - effectiveReserveTokens(ctxWindow, compactionSettings) + : Number.POSITIVE_INFINITY; + const projected = this.#projectSnapcompactContextTokens(preparation, snapcompactResult); + if (projected > budget) { + logger.warn("Snapcompact still overflows the window after frame-budget sizing", { + model: this.#model?.id, + projected, + budget, + }); + snapcompactBlocker = + "snapcompact could not bring the context under the limit; using context-full auto-compaction instead."; + snapcompactResult = undefined; + } + } + } + } + if (snapcompactBlocker) { + this.#host.emitNotice("warning", snapcompactBlocker, "compaction"); + action = "context-full"; + } + } + + if (compactionPrep.kind === "fromHook") { + summary = compactionPrep.summary; + shortSummary = compactionPrep.shortSummary; + firstKeptEntryId = compactionPrep.firstKeptEntryId; + tokensBefore = compactionPrep.tokensBefore; + details = compactionPrep.details; + preserveData = compactionPrep.preserveData; + } else if (snapcompactResult) { + summary = snapcompactResult.summary; + shortSummary = snapcompactResult.shortSummary; + firstKeptEntryId = snapcompactResult.firstKeptEntryId; + tokensBefore = snapcompactResult.tokensBefore; + details = snapcompactResult.details; + preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) }; + } else { + const candidates = this.#getCompactionModelCandidates(availableModels); + const retrySettings = this.#host.settings.getGroup("retry"); + const telemetry = resolveTelemetry(this.#host.agent.telemetry, this.#host.sessionId()); + let compactResult: CompactionResult | undefined; + let lastError: unknown; + codexCompaction = createCodexCompactionContext({ + trigger: "auto", + reason: "context_limit", + phase: + options.phase ?? + (reason === "threshold" ? "pre_turn" : reason === "idle" ? "standalone_turn" : "mid_turn"), + }); + + for (let candidateIndex = 0; candidateIndex < candidates.length; candidateIndex++) { + const candidate = candidates[candidateIndex]; + const hasMoreCandidates = candidateIndex < candidates.length - 1; + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId()); + if (!apiKey) continue; + + let attempt = 0; + while (true) { + try { + compactResult = await compact( + this.#host.obfuscatePreparationForProvider(preparation), + candidate, + this.#host.modelRegistry.resolver(candidate, this.#host.sessionId()), + undefined, + autoCompactionSignal, + { + promptOverride: this.#host.obfuscateTextForProvider(compactionPrep.hookPrompt), + extraContext: compactionPrep.hookContext, + remoteInstructions: this.#host.baseSystemPrompt().join("\n\n"), + metadata: this.#host.agent.metadataForProvider(candidate.provider), + initiatorOverride: "agent", + convertToLlm: messages => this.#host.convertToLlmForSideRequest(messages), + telemetry, + // Honor the user's /model thinking selection on the + // auto-compaction path — the most-fired compaction + // site. Clamped per-model inside compact() via + // resolveCompactionEffort. + thinkingLevel: this.#host.thinkingLevel(), + tools: this.#host.agent.state.tools, + sessionId: this.#host.sessionId(), + promptCacheKey: this.#host.sessionId(), + providerSessionState: this.#host.providerSessionState, + codexCompaction, + }, + ); + break; + } catch (error) { + if (autoCompactionSignal.aborted) { + throw error; + } + + const message = error instanceof Error ? error.message : String(error); + const id = AIError.classify(error, candidate.api); + if (AIError.is(id, AIError.Flag.AuthFailed)) { + lastError = this.#buildCompactionAuthError(); + break; + } + if (AIError.is(id, AIError.Flag.Timeout)) { + logger.warn( + hasMoreCandidates + ? "Auto-compaction summarization timed out, trying next model" + : "Auto-compaction summarization timed out, not retrying same model", + { + error: message, + model: `${candidate.provider}/${candidate.id}`, + }, + ); + lastError = error; + break; + } + + const retryAfterMs = this.#host.parseRetryAfterMsFromError(message); + const shouldRetry = + retrySettings.enabled && + attempt < retrySettings.maxRetries && + (retryAfterMs !== undefined || + AIError.is(id, AIError.Flag.Transient) || + AIError.is(id, AIError.Flag.UsageLimit)); + if (!shouldRetry) { + lastError = error; + break; + } + + const baseDelayMs = retrySettings.baseDelayMs * 2 ** attempt; + const delayMs = retryAfterMs !== undefined ? Math.max(baseDelayMs, retryAfterMs) : baseDelayMs; + + // If retry delay is too long (>30s), try next candidate instead of waiting + const maxAcceptableDelayMs = 30_000; + if (delayMs > maxAcceptableDelayMs && hasMoreCandidates) { + logger.warn("Auto-compaction retry delay too long, trying next model", { + delayMs, + retryAfterMs, + error: message, + model: `${candidate.provider}/${candidate.id}`, + }); + lastError = error; + break; // Exit retry loop, continue to next candidate + } + + attempt++; + logger.warn("Auto-compaction failed, retrying", { + attempt, + maxRetries: retrySettings.maxRetries, + delayMs, + retryAfterMs, + error: message, + model: `${candidate.provider}/${candidate.id}`, + }); + await scheduler.wait(delayMs, { signal: autoCompactionSignal }); + } + } + + if (compactResult) { + break; + } + } + + if (!compactResult) { + if (lastError) { + throw lastError; + } + throw new Error("Compaction failed: no available model"); + } + + summary = compactResult.summary; + shortSummary = compactResult.shortSummary; + firstKeptEntryId = compactResult.firstKeptEntryId; + tokensBefore = compactResult.tokensBefore; + details = compactResult.details; + preserveData = mergeLlmCompactionPreserveData(compactionPrep.preserveData, compactResult.preserveData); + } + + if (autoCompactionSignal.aborted) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: true, + willRetry: false, + }); + return COMPACTION_CHECK_NONE; + } + + this.#host.sessionManager.appendCompaction( + summary, + shortSummary, + firstKeptEntryId, + tokensBefore, + details, + fromExtension, + preserveData, + ); + const newEntries = this.#host.sessionManager.getEntries(); + const sessionContext = this.#host.buildDisplaySessionContext(); + this.#host.agent.replaceMessages(sessionContext.messages); + this.#host.rebaseAfterCompaction(); + // Compaction discarded the conversation history that carried the approved + // plan reference. Clear the sent-flag so #buildPlanReferenceMessage re-reads + // the plan from disk and re-injects it on the next turn (issue #1246). + this.#host.resetPlanReference(); + this.#host.resetAdvisorRuntimes(); + this.#host.syncTodoPhasesFromBranch(); + if (codexCompaction) { + this.#host.resetCodexProviderAfterCompaction(codexCompaction); + } else { + this.#host.closeCodexProviderSessionsForHistoryRewrite(); + } + + // Get the saved compaction entry for the hook + const savedCompactionEntry = newEntries.find(e => e.type === "compaction" && e.summary === summary) as + | CompactionEntry + | undefined; + + if (this.#host.extensionRunner && savedCompactionEntry) { + await this.#host.extensionRunner.emit({ + type: "session_compact", + compactionEntry: savedCompactionEntry, + fromExtension, + }); + } + + const result: CompactionResult = { + summary, + shortSummary, + firstKeptEntryId, + tokensBefore, + details, + preserveData, + }; + // Post-maintenance progress guard — evaluated BEFORE emitting + // auto_compaction_end so the TUI rebuild triggered by that event + // already reflects any rescue rewrite (elide / image-drop) and the + // dead-end warning stamped on the compaction entry. Snapcompact can + // project over budget and fall back to a context-full summary; the + // summarizer keeps `keepRecentTokens` of recent history verbatim and + // findCutPoint can only cut at turn boundaries (never tool results), + // so a single oversized recent turn (e.g. a huge tool result) leaves + // the rewritten context still above threshold. Scheduling the + // continuation regardless means the next agent_end re-enters + // checkCompaction over the same oversized tail and re-fires forever. + // The retry and the threshold auto-continue use different progress + // tests (a recoverable overflow only has to fit; the auto-continue + // thrash needs the stricter recovery band), so each branch evaluates + // its own below. + let continuationScheduled = false; + // A non-idle pass that wanted to continue (retry or auto-continue) but freed + // too little for that path to proceed is a dead-end: warn once so the user + // understands why maintenance paused instead of silently looping. + let noProgressDeadEnd = false; + let retryFits = false; + let hasHeadroom = false; + + if (willRetry) { + const messages = this.#host.agent.state.messages; + const lastMsg = messages[messages.length - 1]; + if (lastMsg?.role === "assistant") { + const lastAssistant = lastMsg as AssistantMessage; + // Drop the prior turn before retry when it carries no actionable deliverable: + // - "error": failure was kept in history but must not re-enter the next turn's prompt. + // - reason === "incomplete" && stopReason === "length": truncated output (typically + // reasoning-only) — re-running it produces the same dead-end. + const shouldDrop = + lastAssistant.stopReason === "error" || + (reason === "incomplete" && lastAssistant.stopReason === "length"); + if (shouldDrop) { + this.#host.agent.replaceMessages(messages.slice(0, -1)); + this.#host.rebaseAfterCompaction(); + } + } + + // Retry only needs the rebuilt prompt to fit the window again — measured + // AFTER the drop above so the just-failed turn (which the retry prompt + // won't include) is excluded. Reusing the auto-continue recovery band + // here turned recoverable overflows into manual dead-ends (#3412 review), + // so use the looser fit budget. + retryFits = this.#compactionCreatedRetryFit(); + if (!retryFits) { + retryFits = await this.#rescueCompactionDeadEnd(autoCompactionSignal, { + skipElide: fallbackFromShake, + hasProgress: () => this.#compactionCreatedRetryFit(), + }); + } + if (!retryFits) { + noProgressDeadEnd = true; + } + } else if (reason !== "idle") { + // Mirror the shake recovery-band check: only auto-continue when compaction + // landed residual context under `COMPACTION_RECOVERY_BAND × threshold`. + // Re-firing on a history that still sits just over the line is the + // snapcompact thrash, so require genuine headroom, not a bare fit. Even + // when auto-continue is disabled, a no-headroom threshold pass must still + // block later automatic continuations (todo reminders/session_stop hooks) + // from re-entering the same oversized context. + hasHeadroom = this.#compactionCreatedHeadroom(); + if (!hasHeadroom) { + hasHeadroom = await this.#rescueCompactionDeadEnd(autoCompactionSignal, { + skipElide: fallbackFromShake, + hasProgress: () => this.#compactionCreatedHeadroom(), + }); + } + if (!hasHeadroom) { + noProgressDeadEnd = true; + } + } + + const deadEndWarning = noProgressDeadEnd ? compactionDeadEndWarning("clear large tool output") : undefined; + if (deadEndWarning) { + // Stamp the divider: the compaction bar badges the dead-end and + // carries the full warning in its ctrl+o detail, so the pause + // stays explained even after the notice row scrolls away. Stamp + // the branch's LATEST compaction entry — a frame rescue may have + // superseded `savedCompactionEntry` with a rebuilt one, and the + // collapsed transcript badges only the active entry. + const stampEntry = getLatestCompactionEntry(this.#host.sessionManager.getBranch()) ?? savedCompactionEntry; + if (stampEntry) { + stampEntry.warning = deadEndWarning; + await this.#host.sessionManager.rewriteEntries(); + } + } + + await this.#host.emitSessionEvent({ type: "auto_compaction_end", action, result, aborted: false, willRetry }); + + if (retryFits) { + this.#host.scheduleAgentContinue({ delayMs: 100, generation }); + continuationScheduled = true; + } else { + continuationScheduled = this.#host.scheduleCompactionContinuation({ + generation, + autoContinue: hasHeadroom && shouldAutoContinue, + terminalTextAnswer, + suppressContinuation, + }); + } + + if (deadEndWarning) { + this.#host.emitNotice("warning", deadEndWarning, "compaction"); + } + if (continuationScheduled) return COMPACTION_CHECK_CONTINUATION; + return noProgressDeadEnd ? COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION : COMPACTION_CHECK_NONE; + } catch (error) { + if (autoCompactionSignal.aborted) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: true, + willRetry: false, + }); + return COMPACTION_CHECK_NONE; + } + const errorMessage = error instanceof Error ? error.message : "compaction failed"; + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + errorMessage: + reason === "overflow" + ? `Context overflow recovery failed: ${errorMessage}` + : reason === "incomplete" + ? `Incomplete response recovery failed: ${errorMessage}` + : `Auto-compaction failed: ${errorMessage}`, + }); + } finally { + if (this.#autoCompactionAbortController === autoCompactionAbortController) { + this.#autoCompactionAbortController = undefined; + } + } + return COMPACTION_CHECK_NONE; + } + + /** + * Run a shake-strategy auto-maintenance pass. Emits the + * `auto_compaction_start`/`auto_compaction_end` pair with a shake `action`, + * runs {@link shake} inline against the protect-window config, and schedules + * continuation exactly like the context-full tail. + * + * Returns `"fallback"` only for an overflow recovery where shake reclaimed + * nothing (or threw) — the caller then runs the summary-compaction body so + * the oversized input still gets resolved. Returns `"handled"` otherwise. + */ + async #runAutoShake( + reason: "overflow" | "threshold" | "idle" | "incomplete", + willRetry: boolean, + generation: number, + autoContinue: boolean, + terminalTextAnswer: boolean, + triggerContextTokens?: number, + suppressContinuation = false, + ): Promise { + const action = "shake"; + this.#autoCompactionAbortController?.abort(); + const controller = new AbortController(); + this.#autoCompactionAbortController = controller; + const signal = controller.signal; + try { + await this.#host.emitSessionEvent({ type: "auto_compaction_start", reason, action }); + const result = await this.#host.shake("elide", { config: DEFAULT_SHAKE_CONFIG, signal }); + if (signal.aborted) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: true, + willRetry: false, + }); + return COMPACTION_CHECK_NONE; + } + const reclaimed = result.toolResultsDropped + result.blocksDropped > 0; + // Detect the dead-loop reported in issues #2119/#2275: the threshold check + // fires, shake runs, but residual context is still above the configured + // threshold. The next agent_end would re-trigger shake, which has nothing + // new to drop on the second pass, so the loop spins until the user kills it. + // Same hazard for "incomplete" (the retry would re-hit the length cap) and + // for the existing "overflow + nothing reclaimed" case. In every recovery + // reason we hand off to the summarization-driven context-full path so the + // situation actually resolves; "idle" is exempt because its 60s+ timer + // re-checks usage before re-firing and cannot dead-loop on its own. + // + // #2275: the post-shake check MUST stay provider-anchored when caller + // usage and local estimates diverge. The local estimator undercounts + // thinking-signature payloads, so thinking-heavy sessions can read well + // below the provider usage that fired the threshold. Prefer the caller's + // context figure when supplied, then subtract shake's own savings and add + // hysteresis (80% recovery band) so we don't oscillate at the boundary. + // Threshold callers pass the provider-billed trigger after accounting for + // any supersede/drop-useless pruning that already rewrote the next prompt; + // without that pre-shake savings, shake can fall through to context-full + // even though the post-prune history is already inside the recovery band. + const contextWindow = this.#model?.contextWindow ?? 0; + const compactionSettings = this.#host.settings.getGroup("compaction"); + let stillOverThreshold = false; + if (contextWindow > 0) { + if (typeof triggerContextTokens === "number" && Number.isFinite(triggerContextTokens)) { + const correctedTokens = Math.max(0, triggerContextTokens - result.tokensFreed); + const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); + const recoveryBand = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); + stillOverThreshold = correctedTokens > recoveryBand; + } else { + const postShakeTokens = this.#host.getContextUsage({ contextWindow })?.tokens ?? 0; + stillOverThreshold = shouldCompact(postShakeTokens, contextWindow, compactionSettings); + } + } + const shouldFallBack = reason !== "idle" && ((reason === "overflow" && !reclaimed) || stillOverThreshold); + if (shouldFallBack) { + const errorMessage = reclaimed + ? `Auto-shake reclaimed ~${result.tokensFreed} tokens but context is still above the threshold; falling back to context-full compaction.` + : "Auto-shake found nothing eligible to drop; falling back to context-full compaction."; + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + skipped: !reclaimed, + errorMessage, + }); + return "fallback"; + } + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry, + skipped: !reclaimed, + }); + + let continuationScheduled = false; + if (willRetry) { + // The shake rebuild replays every entry, so a trailing error/length + // assistant from the failed turn re-enters agent state — drop it before + // retrying, same as the context-full tail. + const messages = this.#host.agent.state.messages; + const lastMsg = messages[messages.length - 1]; + if (lastMsg?.role === "assistant") { + const lastAssistant = lastMsg as AssistantMessage; + const shouldDrop = + lastAssistant.stopReason === "error" || + (reason === "incomplete" && lastAssistant.stopReason === "length"); + if (shouldDrop) this.#host.agent.replaceMessages(messages.slice(0, -1)); + } + this.#host.scheduleAgentContinue({ delayMs: 100, generation }); + continuationScheduled = true; + } else { + continuationScheduled = this.#host.scheduleCompactionContinuation({ + generation, + autoContinue: reason !== "idle" && autoContinue, + terminalTextAnswer, + suppressContinuation, + }); + } + if (!reclaimed) { + return willRetry && continuationScheduled + ? { ...COMPACTION_CHECK_CONTINUATION, historyRewritten: true } + : continuationScheduled + ? COMPACTION_CHECK_CONTINUATION + : COMPACTION_CHECK_NONE; + } + return { + ...(continuationScheduled ? COMPACTION_CHECK_CONTINUATION : COMPACTION_CHECK_NONE), + historyRewritten: true, + }; + } catch (error) { + if (signal.aborted) { + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: true, + willRetry: false, + }); + return COMPACTION_CHECK_NONE; + } + const message = error instanceof Error ? error.message : "shake failed"; + await this.#host.emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + errorMessage: message, + skipped: false, + }); + // Overflow still needs recovery even if shake threw. + return reason === "overflow" ? "fallback" : COMPACTION_CHECK_NONE; + } finally { + if (this.#autoCompactionAbortController === controller) { + this.#autoCompactionAbortController = undefined; + } + } + } + + /** + * Toggle auto-compaction setting. + */ + setAutoCompactionEnabled(enabled: boolean): void { + this.#host.settings.set("compaction.enabled", enabled); + if (enabled && this.#host.settings.get("compaction.strategy") === "off") { + const defaultStrategy = getDefault("compaction.strategy"); + this.#host.settings.set("compaction.strategy", defaultStrategy === "off" ? "context-full" : defaultStrategy); + } + } + + /** Whether auto-compaction is enabled */ + get autoCompactionEnabled(): boolean { + return this.#host.settings.get("compaction.enabled") && this.#host.settings.get("compaction.strategy") !== "off"; + } +} diff --git a/packages/coding-agent/src/session/session-memory.ts b/packages/coding-agent/src/session/session-memory.ts new file mode 100644 index 000000000..d89165458 --- /dev/null +++ b/packages/coding-agent/src/session/session-memory.ts @@ -0,0 +1,222 @@ +/** Session memory backend lifecycle and transcript resets. */ + +import type { Agent, AgentTool } from "@oh-my-pi/pi-agent-core"; +import { logger } from "@oh-my-pi/pi-utils"; +import type { ModelRegistry } from "../config/model-registry"; +import type { Settings } from "../config/settings"; +import type { HindsightSessionState } from "../hindsight/state"; +import { resolveMemoryBackend } from "../memory-backend/resolve"; +import type { MemoryBackendStartOptions } from "../memory-backend/types"; +import type { MnemopiSessionState } from "../mnemopi/state"; + +/** Capabilities borrowed from the owning AgentSession. */ +export interface SessionMemoryHost { + agent: Agent; + settings: Settings; + modelRegistry: ModelRegistry; + isDisposed(): boolean; + memoryBackendSession(): MemoryBackendStartOptions["session"]; + getHindsightSessionState(): HindsightSessionState | undefined; + setHindsightSessionState(state: HindsightSessionState | undefined): void; + getMnemopiSessionState(): MnemopiSessionState | undefined; + takeMnemopiSessionState(): MnemopiSessionState | undefined; + setBaseSystemPrompt(prompt: string[]): void; + refreshBaseSystemPrompt(): Promise; + replaceMemoryTools(tools: AgentTool[]): Promise; +} + +/** Owns memory backend transitions and transcript-scoped memory state. */ +export class SessionMemory { + readonly #host: SessionMemoryHost; + readonly #memoryAgentDir: string | undefined; + readonly #memoryTaskDepth: number; + readonly #createMemoryTools: (() => Promise) | undefined; + #memoryBackendTransition: Promise = Promise.resolve(); + #localMemoryStartupAbort: AbortController | undefined; + #baseSystemPromptBeforeMemoryPromotion: string[] | undefined; + + constructor( + host: SessionMemoryHost, + options: { + memoryAgentDir?: string; + memoryTaskDepth?: number; + createMemoryTools?: () => Promise; + }, + ) { + this.#host = host; + this.#memoryAgentDir = options.memoryAgentDir; + this.#memoryTaskDepth = options.memoryTaskDepth ?? 0; + this.#createMemoryTools = options.createMemoryTools; + } + + /** Current serialized backend transition, used by prompt and disposal drains. */ + get transition(): Promise { + return this.#memoryBackendTransition; + } + + /** Base prompt captured before a per-turn memory promotion. */ + get promotionSnapshot(): string[] | undefined { + return this.#baseSystemPromptBeforeMemoryPromotion; + } + + /** Clears the per-turn memory promotion after a canonical prompt rebuild. */ + clearPromotionSnapshot(): void { + this.#baseSystemPromptBeforeMemoryPromotion = undefined; + } + + /** Captures the canonical prompt before the first per-turn memory promotion. */ + capturePromotionSnapshot(prompt: string[]): void { + this.#baseSystemPromptBeforeMemoryPromotion ??= prompt; + } + + /** Restores a promotion snapshot while rolling back a failed session switch. */ + restorePromotionSnapshot(prompt: string[] | undefined): void { + this.#baseSystemPromptBeforeMemoryPromotion = prompt; + } + /** Rekeys every active memory backend to the current provider session. */ + rekeyForCurrentSessionId(): void { + this.#rekeyHindsightMemoryForCurrentSessionId(); + this.#rekeyMnemopiMemoryForCurrentSessionId(); + } + + #rekeyHindsightMemoryForCurrentSessionId(): void { + if (this.#host.settings.get("memory.backend") !== "hindsight") return; + const sid = this.#host.agent.sessionId; + if (!sid) return; + this.#host.getHindsightSessionState()?.setSessionId(sid); + } + + #rekeyMnemopiMemoryForCurrentSessionId(): void { + if (this.#host.settings.get("memory.backend") !== "mnemopi") return; + const sid = this.#host.agent.sessionId; + if (!sid) return; + this.#host.getMnemopiSessionState()?.setSessionId(sid); + } + + /** New session file: reset auto-recall / retain-threshold counters for the new transcript. */ + #resetHindsightConversationTrackingIfHindsight(): boolean { + if (this.#host.settings.get("memory.backend") !== "hindsight") return false; + const state = this.#host.getHindsightSessionState(); + if (!state || state.aliasOf) return false; + state.resetConversationTracking(); + return true; + } + + #resetMnemopiConversationTrackingIfMnemopi(): boolean { + if (this.#host.settings.get("memory.backend") !== "mnemopi") return false; + const state = this.#host.getMnemopiSessionState(); + if (!state || state.aliasOf) return false; + state.resetConversationTracking(); + return true; + } + + /** Resets transcript-scoped memory counters and removes a promoted prompt. */ + async resetContextForNewTranscript(): Promise { + const hadPromotedMemoryPrompt = this.#baseSystemPromptBeforeMemoryPromotion !== undefined; + const resetHindsight = this.#resetHindsightConversationTrackingIfHindsight(); + const resetMnemopi = this.#resetMnemopiConversationTrackingIfMnemopi(); + if (hadPromotedMemoryPrompt) { + this.#host.setBaseSystemPrompt(this.#baseSystemPromptBeforeMemoryPromotion!); + this.#baseSystemPromptBeforeMemoryPromotion = undefined; + } + if (resetHindsight || resetMnemopi || hadPromotedMemoryPrompt) { + await this.#host.refreshBaseSystemPrompt(); + } + } + + /** Cancel the local rollout-memory startup owned by this session. */ + cancelLocalMemoryStartup(): void { + this.#localMemoryStartupAbort?.abort(); + this.#localMemoryStartupAbort = undefined; + } + + /** Start a new local rollout-memory generation and cancel its predecessor. */ + beginLocalMemoryStartup(): AbortSignal { + this.cancelLocalMemoryStartup(); + const controller = new AbortController(); + this.#localMemoryStartupAbort = controller; + return controller.signal; + } + + /** Release the local startup slot if `signal` still owns it. */ + endLocalMemoryStartup(signal: AbortSignal): void { + if (this.#localMemoryStartupAbort?.signal === signal) this.#localMemoryStartupAbort = undefined; + } + + async #disposeMemoryBackendState(consolidateMnemopi = true): Promise { + this.cancelLocalMemoryStartup(); + const hindsight = this.#host.getHindsightSessionState(); + if (hindsight) { + try { + await hindsight.flushRetainQueue(); + } catch (error) { + logger.warn("Memory lifecycle: Hindsight flush failed", { error: String(error) }); + } + this.#host.setHindsightSessionState(undefined); + hindsight.dispose(); + } + + const mnemopi = this.#host.takeMnemopiSessionState(); + if (mnemopi) { + try { + await mnemopi.dispose({ consolidate: consolidateMnemopi }); + } catch (error) { + logger.warn("Memory lifecycle: Mnemopi dispose failed", { error: String(error) }); + } + } + } + + /** + * Apply the selected memory backend to runtime state, tools, and prompt. + * Concurrent settings changes run in order and settle before the next turn. + */ + async applyMemoryBackend(): Promise { + if (this.#host.isDisposed()) return; + const transition = this.#memoryBackendTransition.then(() => this.#applyMemoryBackend()); + this.#memoryBackendTransition = transition.then( + () => undefined, + () => undefined, + ); + await transition; + } + + async #applyMemoryBackend(): Promise { + if (this.#host.isDisposed()) return; + try { + await this.#disposeMemoryBackendState(); + if (this.#memoryAgentDir && this.#memoryTaskDepth === 0 && !this.#host.isDisposed()) { + const backend = await resolveMemoryBackend(this.#host.settings); + await backend.start({ + session: this.#host.memoryBackendSession(), + settings: this.#host.settings, + modelRegistry: this.#host.modelRegistry, + agentDir: this.#memoryAgentDir, + taskDepth: this.#memoryTaskDepth, + }); + } + if (this.#host.isDisposed()) return; + await this.#refreshMemoryTools(); + if (this.#host.isDisposed()) return; + await this.#host.refreshBaseSystemPrompt(); + } catch (error) { + await this.#disposeMemoryBackendState(false); + if (!this.#host.isDisposed()) { + await this.#replaceMemoryTools([]).catch(refreshError => { + logger.warn("Failed to remove memory tools after backend apply error", { + error: String(refreshError), + }); + }); + } + throw error; + } + } + + async #refreshMemoryTools(): Promise { + const tools = (await this.#createMemoryTools?.()) ?? []; + await this.#replaceMemoryTools(tools); + } + + #replaceMemoryTools(tools: AgentTool[]): Promise { + return this.#host.replaceMemoryTools(tools); + } +} diff --git a/packages/coding-agent/src/session/session-provider-boundary.ts b/packages/coding-agent/src/session/session-provider-boundary.ts new file mode 100644 index 000000000..cbd97a970 --- /dev/null +++ b/packages/coding-agent/src/session/session-provider-boundary.ts @@ -0,0 +1,307 @@ +/** Provider-facing message, image, secret, and stream normalization for a session. */ + +import type { Agent, AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { CompactionPreparation } from "@oh-my-pi/pi-agent-core/compaction"; +import type { AssistantMessage, ImageContent, Message, Model, SimpleStreamOptions, TextContent } from "@oh-my-pi/pi-ai"; +import { isRecord, logger } from "@oh-my-pi/pi-utils"; +import * as snapcompact from "@oh-my-pi/snapcompact"; +import type { ModelRegistry } from "../config/model-registry"; +import { formatModelString } from "../config/model-resolver"; +import type { Settings } from "../config/settings"; +import { validateProviderMaxInFlightRequests } from "../config/settings"; +import type { LocalProtocolOptions } from "../internal-urls"; +import { + deobfuscateSessionContext, + obfuscateMessages, + type SecretObfuscator, + stripPendingSecretPlaceholderSuffix, +} from "../secrets/obfuscator"; +import { normalizeModelContextImages } from "../utils/image-loading"; +import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback"; +import { type CustomMessage, convertToLlm } from "./messages"; +import { IMAGE_ATTACHMENT_DESCRIPTION_TYPE } from "./queued-messages"; +import type { BuildSessionContextOptions, SessionContext } from "./session-context"; +import type { SessionManager } from "./session-manager"; + +type NormalizableContentBlock = AssistantMessage["content"][number] | TextContent | ImageContent; + +/** Capabilities borrowed from the owning AgentSession. */ +export interface SessionProviderBoundaryHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + modelRegistry: ModelRegistry; + model(): Model | undefined; + sessionId(): string; + localProtocolOptions(): LocalProtocolOptions; + transformContext(messages: AgentMessage[], signal?: AbortSignal): AgentMessage[] | Promise; + convertToLlm(messages: AgentMessage[]): Message[] | Promise; + onPayload: SimpleStreamOptions["onPayload"] | undefined; + onResponse: SimpleStreamOptions["onResponse"] | undefined; + onSseEvent: SimpleStreamOptions["onSseEvent"] | undefined; + obfuscator: SecretObfuscator | undefined; +} + +/** Owns the transformations at the session/provider boundary. */ +export class SessionProviderBoundary { + readonly #host: SessionProviderBoundaryHost; + + constructor(host: SessionProviderBoundaryHost) { + this.#host = host; + } + + /** Latest image attachments addressable by tools as `Image #N` or `attachment://N`. */ + getImageAttachments(): { label: string; uri: string; image: ImageContent }[] { + for (let i = this.#host.agent.state.messages.length - 1; i >= 0; i--) { + const message = this.#host.agent.state.messages[i]; + if (!message || (message.role !== "user" && message.role !== "developer") || !Array.isArray(message.content)) { + continue; + } + const images = message.content.filter((part): part is ImageContent => part.type === "image"); + if (images.length === 0) continue; + return images.map((image, index) => ({ + label: `Image #${index + 1}`, + uri: `attachment://${index + 1}`, + image, + })); + } + return []; + } + + /** Builds the current deobfuscated context for agent display and replay. */ + buildDisplaySessionContext(): SessionContext { + return deobfuscateSessionContext(this.#host.sessionManager.buildSessionContext(), this.#host.obfuscator); + } + + /** Builds the full display-only transcript context. */ + buildTranscriptSessionContext( + options?: Pick, + ): SessionContext { + return deobfuscateSessionContext( + this.#host.sessionManager.buildSessionContext({ + transcript: true, + collapseCompactedHistory: options?.collapseCompactedHistory, + keepDanglingToolCalls: options?.keepDanglingToolCalls, + }), + this.#host.obfuscator, + true, + ); + } + + /** Obfuscates optional plaintext before a provider request. */ + obfuscateText(text: string | undefined): string | undefined { + if (!text || !this.#host.obfuscator?.hasSecrets()) return text; + return this.#host.obfuscator.obfuscate(text); + } + + /** Obfuscates summaries and snapcompact plaintext carried into compaction. */ + obfuscateCompactionPreparation(preparation: CompactionPreparation): CompactionPreparation { + if (!this.#host.obfuscator?.hasSecrets()) return preparation; + const previousSummary = this.obfuscateText(preparation.previousSummary); + const previousPreserveData = this.#obfuscatePreservedArchiveText(preparation.previousPreserveData); + if ( + previousSummary === preparation.previousSummary && + previousPreserveData === preparation.previousPreserveData + ) { + return preparation; + } + return { ...preparation, previousSummary, previousPreserveData }; + } + + /** Deobfuscates provider text before exposing it to the session. */ + deobfuscateText(text: string): string { + if (!this.#host.obfuscator?.hasSecrets()) return text; + return this.#host.obfuscator.deobfuscate(text); + } + + /** Deobfuscates a streamed delta and removes an incomplete secret placeholder suffix. */ + deobfuscateDelta(text: string): string { + const deobfuscated = this.deobfuscateText(text); + if (!this.#host.obfuscator?.hasSecrets()) return deobfuscated; + return stripPendingSecretPlaceholderSuffix(deobfuscated); + } + + /** Converts side-request messages through the session's secret boundary. */ + convertToLlmForSideRequest(messages: AgentMessage[]): Message[] { + const converted = convertToLlm(messages); + return this.#host.obfuscator?.hasSecrets() ? obfuscateMessages(this.#host.obfuscator, converted) : converted; + } + + /** Converts session messages using the configured pre-LLM pipeline. */ + async convertMessagesToLlm(messages: AgentMessage[], signal?: AbortSignal): Promise { + const transformedMessages = await this.#host.transformContext(messages, signal); + return await this.#host.convertToLlm(transformedMessages); + } + + /** Applies session-level stream hooks and provider defaults to a side request. */ + prepareSimpleStreamOptions(options: SimpleStreamOptions, provider = "anthropic"): SimpleStreamOptions { + const sessionOnPayload = this.#host.onPayload; + const sessionOnResponse = this.#host.onResponse; + const sessionMetadata = this.#host.agent.metadataForProvider(provider); + const sessionOnSseEvent = this.#host.onSseEvent; + const openrouterRoutingPreset = + provider === "openrouter" ? this.#host.settings.get("providers.openrouterVariant") : "default"; + const openrouterVariant = + openrouterRoutingPreset !== "default" && options.openrouterVariant === undefined + ? openrouterRoutingPreset + : undefined; + const antigravityEndpointMode = + provider === "google-antigravity" ? this.#host.settings.get("providers.antigravityEndpoint") : undefined; + + const preparedOptions: SimpleStreamOptions = { + ...options, + ...(openrouterVariant !== undefined && { openrouterVariant }), + ...(antigravityEndpointMode !== undefined && { antigravityEndpointMode }), + maxInFlightRequests: validateProviderMaxInFlightRequests( + options.maxInFlightRequests ?? this.#host.settings.get("providers.maxInFlightRequests"), + ), + loopGuard: { + enabled: this.#host.settings.get("model.loopGuard.enabled"), + checkAssistantContent: this.#host.settings.get("model.loopGuard.checkAssistantContent"), + ...options.loopGuard, + }, + }; + + if (sessionMetadata && !options.metadata) { + preparedOptions.metadata = sessionMetadata; + } + + if (sessionOnPayload) { + if (!options.onPayload) { + preparedOptions.onPayload = sessionOnPayload; + } else { + const requestOnPayload = options.onPayload; + preparedOptions.onPayload = async (payload, model) => { + const sessionPayload = await sessionOnPayload(payload, model); + const sessionResolvedPayload = sessionPayload ?? payload; + const requestPayload = await requestOnPayload(sessionResolvedPayload, model); + return requestPayload ?? sessionResolvedPayload; + }; + } + } + + if (sessionOnResponse) { + if (!options.onResponse) { + preparedOptions.onResponse = sessionOnResponse; + } else { + const requestOnResponse = options.onResponse; + preparedOptions.onResponse = async (response, model) => { + await sessionOnResponse(response, model); + await requestOnResponse(response, model); + }; + } + } + + if (sessionOnSseEvent) { + if (!options.onSseEvent) { + preparedOptions.onSseEvent = sessionOnSseEvent; + } else { + const requestOnSseEvent = options.onSseEvent; + preparedOptions.onSseEvent = (event, model) => { + sessionOnSseEvent(event, model); + requestOnSseEvent(event, model); + }; + } + } + + return preparedOptions; + } + + /** Normalizes image payloads for the active model. */ + normalizeImagesForModel(images: ImageContent[] | undefined): Promise { + return normalizeModelContextImages(images, { model: this.#host.model() }); + } + + /** Builds a hidden vision-model description for attachments sent to a text-only model. */ + async buildImageDescriptionNotice( + normalizedImages: ImageContent[], + signal?: AbortSignal, + ): Promise { + const model = this.#host.model(); + const shouldDescribe = + !!model && + !model.input.includes("image") && + !this.#host.settings.get("images.blockImages") && + this.#host.settings.get("images.describeForTextModels"); + if (!shouldDescribe || !model) return undefined; + + let blocks: TextContent[]; + try { + blocks = await describeAttachedImagesForTextModel( + normalizedImages, + { + activeModel: model, + modelRegistry: this.#host.modelRegistry, + settings: this.#host.settings, + localProtocolOptions: this.#host.localProtocolOptions(), + activeModelString: formatModelString(model), + telemetryConfig: this.#host.agent.telemetry, + sessionId: this.#host.sessionId(), + }, + signal, + ); + } catch (error) { + logger.warn("image attachment vision fallback failed; image left undescribed", { + error: error instanceof Error ? error.message : String(error), + }); + return undefined; + } + if (blocks.length === 0) return undefined; + return { + role: "custom", + customType: IMAGE_ATTACHMENT_DESCRIPTION_TYPE, + content: blocks, + display: false, + attribution: "user", + timestamp: Date.now(), + }; + } + + /** Normalizes every image embedded in an agent message. */ + async normalizeAgentMessageImages(message: T): Promise { + if (!("content" in message)) return message; + const content = message.content; + if (typeof content !== "string" && !Array.isArray(content)) return message; + const normalized = await this.#normalizeMessageContentImages(content); + if (normalized === content) return message; + return Object.assign({}, message, { content: normalized }); + } + + async #normalizeMessageContentImages( + content: string | NormalizableContentBlock[], + ): Promise { + if (typeof content === "string") return content; + const images = content.filter((part): part is ImageContent => part.type === "image"); + if (images.length === 0) return content; + const normalizedImages = await this.normalizeImagesForModel(images); + if (!normalizedImages) return content; + let imageIndex = 0; + return content.map(part => (part.type === "image" ? normalizedImages[imageIndex++]! : part)); + } + + #obfuscatePreservedArchiveText( + preserveData: Record | undefined, + ): Record | undefined { + const obfuscator = this.#host.obfuscator; + const slot = preserveData?.[snapcompact.PRESERVE_KEY]; + if ( + !obfuscator?.hasSecrets() || + !preserveData || + !isRecord(slot) || + !snapcompact.getPreservedArchive(preserveData) + ) { + return preserveData; + } + const obfuscated: Record = { ...slot }; + let changed = false; + for (const key of ["text", "textHead", "textTail"] as const) { + const value = slot[key]; + if (typeof value !== "string" || value.length === 0) continue; + const next = obfuscator.obfuscate(value); + if (next === value) continue; + obfuscated[key] = next; + changed = true; + } + return changed ? { ...preserveData, [snapcompact.PRESERVE_KEY]: obfuscated } : preserveData; + } +} diff --git a/packages/coding-agent/src/session/session-stats.ts b/packages/coding-agent/src/session/session-stats.ts new file mode 100644 index 000000000..7b4d6483a --- /dev/null +++ b/packages/coding-agent/src/session/session-stats.ts @@ -0,0 +1,293 @@ +import type { Agent, AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { calculatePromptTokens, estimateTokens, type SessionMessageEntry } from "@oh-my-pi/pi-agent-core/compaction"; +import type { AssistantMessage, Model, ProviderResponseMetadata, Usage } from "@oh-my-pi/pi-ai"; +import { isRecord } from "@oh-my-pi/pi-utils"; +import type { ModelRegistry } from "../config/model-registry"; +import type { ContextUsage } from "../extensibility/extensions/types"; +import { + computeNonMessageBreakdown, + computeNonMessageTokens, + type NonMessageTokenSource, +} from "../modes/utils/context-usage"; +import type { ContextUsageBreakdown, SessionStats } from "./agent-session-types"; +import { getLatestCompactionEntry } from "./session-context"; +import type { SessionManager } from "./session-manager"; + +interface PendingContextSnapshot { + promptTokens: number; + nonMessageTokens: number; + cutoffCount: number; +} + +/** Capabilities the stats tracker borrows from its owning session. */ +export interface SessionStatsTrackerHost { + session: NonMessageTokenSource; + agent: Agent; + sessionManager: SessionManager; + modelRegistry: ModelRegistry; + model(): Model | undefined; + sessionId(): string; +} + +/** Computes session totals and tracks the in-flight context estimate. */ +export class SessionStatsTracker { + readonly #host: SessionStatsTrackerHost; + #pendingContextSnapshot: PendingContextSnapshot | undefined; + #contextUsageRevision = 0; + + constructor(host: SessionStatsTrackerHost) { + this.#host = host; + } + + /** Returns aggregate message, token, and cost statistics for the session. */ + getSessionStats(): SessionStats { + const state = this.#host.agent.state; + const userMessages = state.messages.filter(message => message.role === "user").length; + const assistantMessages = state.messages.filter(message => message.role === "assistant").length; + const toolResults = state.messages.filter(message => message.role === "toolResult").length; + let toolCalls = 0; + let totalInput = 0; + let totalOutput = 0; + let totalCacheRead = 0; + let totalReasoning = 0; + let totalCacheWrite = 0; + let totalTokens = 0; + let totalCost = 0; + let totalPremiumRequests = 0; + for (const message of state.messages) { + if (message.role === "assistant") { + const assistant = message; + toolCalls += assistant.content.filter(content => content.type === "toolCall").length; + totalInput += assistant.usage.input; + totalOutput += assistant.usage.output; + totalReasoning += assistant.usage.reasoningTokens ?? 0; + totalCacheRead += assistant.usage.cacheRead; + totalCacheWrite += assistant.usage.cacheWrite; + totalTokens += assistant.usage.totalTokens; + totalPremiumRequests += assistant.usage.premiumRequests ?? 0; + totalCost += assistant.usage.cost.total; + } + if (message.role === "toolResult" && message.toolName === "task") { + const usage = taskToolUsage(message.details); + if (!usage) continue; + totalInput += usage.input; + totalOutput += usage.output; + totalReasoning += usage.reasoningTokens ?? 0; + totalCacheRead += usage.cacheRead; + totalCacheWrite += usage.cacheWrite; + totalTokens += usage.totalTokens; + totalPremiumRequests += usage.premiumRequests ?? 0; + totalCost += usage.cost.total; + } + } + return { + sessionFile: this.#host.sessionManager.getSessionFile(), + sessionId: this.#host.sessionId(), + userMessages, + assistantMessages, + toolCalls, + toolResults, + totalMessages: state.messages.length, + tokens: { + input: totalInput, + output: totalOutput, + reasoning: totalReasoning, + cacheRead: totalCacheRead, + cacheWrite: totalCacheWrite, + total: totalTokens, + }, + cost: totalCost, + premiumRequests: totalPremiumRequests, + contextUsage: this.getContextUsage(), + }; + } + + /** Returns the current provider-context token breakdown. */ + getContextBreakdown(options?: { + contextWindow?: number; + pendingMessages?: AgentMessage[]; + }): ContextUsageBreakdown | undefined { + const rawContextWindow = options?.contextWindow ?? this.#host.model()?.contextWindow ?? 0; + const contextWindow = Number.isFinite(rawContextWindow) && rawContextWindow > 0 ? rawContextWindow : 0; + const { skillsTokens, toolsTokens, systemContextTokens, systemPromptTokens } = computeNonMessageBreakdown( + this.#host.session, + ); + const categoryNonMessageTokens = skillsTokens + toolsTokens + systemContextTokens + systemPromptTokens; + const currentNonMessageTokens = computeNonMessageTokens(this.#host.session); + const branchEntries = this.#host.sessionManager.getBranch(); + const latestCompaction = getLatestCompactionEntry(branchEntries); + const compactionIndex = latestCompaction ? branchEntries.lastIndexOf(latestCompaction) : -1; + let usedTokens = 0; + let anchored = false; + const pendingMessages = options?.pendingMessages ?? []; + const pending = this.#pendingContextSnapshot; + + let anchorEntry: SessionMessageEntry | undefined; + for (let index = branchEntries.length - 1; index > compactionIndex; index--) { + const entry = branchEntries[index]; + if (entry.type !== "message" || entry.message.role !== "assistant") continue; + const assistant = entry.message; + if (assistant.stopReason !== "aborted" && assistant.stopReason !== "error" && assistant.usage) { + anchorEntry = entry; + break; + } + } + + const activeMessages = this.#host.agent.state.messages; + let anchorIndex = -1; + let anchorAssistant: AssistantMessage | undefined; + if (anchorEntry?.message.role === "assistant") { + const assistant = anchorEntry.message; + anchorAssistant = assistant; + anchorIndex = activeMessages.indexOf(assistant); + if (anchorIndex === -1) { + anchorIndex = activeMessages.findIndex( + message => message.role === "assistant" && message.timestamp === assistant.timestamp, + ); + } + } + + const useAnchor = + anchorAssistant !== undefined && anchorIndex !== -1 && (!pending || anchorIndex >= pending.cutoffCount); + if (useAnchor && anchorAssistant) { + const promptTokens = + anchorAssistant.contextSnapshot?.promptTokens ?? calculatePromptTokens(anchorAssistant.usage); + const nonMessageTokens = + anchorAssistant.contextSnapshot?.nonMessageTokens ?? computeNonMessageTokens(this.#host.session); + anchored = true; + let tailTokens = 0; + for (let index = anchorIndex + 1; index < activeMessages.length; index++) { + tailTokens += estimateTokens(activeMessages[index]); + } + usedTokens = + promptTokens + + Math.max(0, currentNonMessageTokens - nonMessageTokens) + + tailTokens + + pendingMessages.reduce((sum, message) => sum + estimateTokens(message), 0); + } else if (pending) { + anchored = true; + let tailTokens = 0; + for (let index = pending.cutoffCount; index < activeMessages.length; index++) { + tailTokens += estimateTokens(activeMessages[index]); + } + usedTokens = + pending.promptTokens + + Math.max(0, currentNonMessageTokens - pending.nonMessageTokens) + + tailTokens + + pendingMessages.reduce((sum, message) => sum + estimateTokens(message), 0); + } + + if (!anchored && !pending && branchEntries.length === 0) { + for (let index = activeMessages.length - 1; index >= 0; index--) { + const message = activeMessages[index]; + if ( + message.role !== "assistant" || + message.stopReason === "aborted" || + message.stopReason === "error" || + !message.usage + ) { + continue; + } + const promptTokens = message.contextSnapshot?.promptTokens ?? calculatePromptTokens(message.usage); + const nonMessageTokens = + message.contextSnapshot?.nonMessageTokens ?? computeNonMessageTokens(this.#host.session); + let tailTokens = 0; + for (let tailIndex = index + 1; tailIndex < activeMessages.length; tailIndex++) { + tailTokens += estimateTokens(activeMessages[tailIndex]); + } + usedTokens = + promptTokens + + Math.max(0, currentNonMessageTokens - nonMessageTokens) + + tailTokens + + pendingMessages.reduce((sum, pendingMessage) => sum + estimateTokens(pendingMessage), 0); + anchored = true; + break; + } + } + if (!anchored) { + let messagesTokens = 0; + for (const message of activeMessages) messagesTokens += estimateTokens(message); + usedTokens = + currentNonMessageTokens + + messagesTokens + + pendingMessages.reduce((sum, message) => sum + estimateTokens(message), 0); + } + return { + contextWindow, + anchored, + usedTokens, + systemPromptTokens, + systemToolsTokens: toolsTokens, + systemContextTokens, + skillsTokens, + messagesTokens: Math.max(0, usedTokens - categoryNonMessageTokens), + }; + } + + /** Returns current context tokens, capacity, and percentage. */ + getContextUsage(options?: { contextWindow?: number }): ContextUsage | undefined { + const breakdown = this.getContextBreakdown(options); + if (!breakdown) return undefined; + return { + tokens: breakdown.usedTokens, + contextWindow: breakdown.contextWindow, + percent: breakdown.contextWindow > 0 ? (breakdown.usedTokens / breakdown.contextWindow) * 100 : 0, + }; + } + + /** Monotonic revision for in-flight context snapshot changes. */ + get revision(): number { + return this.#contextUsageRevision; + } + + /** Non-message token count captured for the active provider request. */ + get pendingNonMessageTokens(): number | undefined { + return this.#pendingContextSnapshot?.nonMessageTokens; + } + + /** Sets or clears the in-flight context snapshot. */ + setPendingSnapshot(snapshot: PendingContextSnapshot | undefined): void { + this.#pendingContextSnapshot = snapshot; + this.#contextUsageRevision++; + } + + /** Recomputes an in-flight snapshot after history is compacted or rewritten. */ + rebaseAfterCompaction(): void { + if (!this.#pendingContextSnapshot) return; + const nonMessageTokens = computeNonMessageTokens(this.#host.session); + const messages = this.#host.agent.state.messages; + this.setPendingSnapshot({ + promptTokens: nonMessageTokens + messages.reduce((sum, message) => sum + estimateTokens(message), 0), + nonMessageTokens, + cutoffCount: messages.length, + }); + } + + /** Records provider usage headers against the active session account. */ + ingestProviderUsageHeaders(response: ProviderResponseMetadata, model?: Model): void { + const provider = model?.provider; + if (!provider) return; + this.#host.modelRegistry.authStorage.ingestUsageHeaders(provider, response.headers, { + sessionId: this.#host.agent.sessionId, + baseUrl: this.#host.modelRegistry.getProviderBaseUrl?.(provider), + }); + } +} + +function taskToolUsage(details: unknown): Usage | undefined { + if (!details || typeof details !== "object") return undefined; + const usage = Reflect.get(details, "usage"); + return isUsage(usage) ? usage : undefined; +} + +function isUsage(value: unknown): value is Usage { + if (!isRecord(value) || !isRecord(value.cost)) return false; + return ( + typeof value.input === "number" && + typeof value.output === "number" && + typeof value.cacheRead === "number" && + typeof value.cacheWrite === "number" && + typeof value.totalTokens === "number" && + typeof value.cost.total === "number" + ); +} diff --git a/packages/coding-agent/src/session/session-tools.ts b/packages/coding-agent/src/session/session-tools.ts new file mode 100644 index 000000000..1880b1ef2 --- /dev/null +++ b/packages/coding-agent/src/session/session-tools.ts @@ -0,0 +1,933 @@ +import type { Agent, AgentTool } from "@oh-my-pi/pi-agent-core"; +import type { Model } from "@oh-my-pi/pi-ai"; +import { logger, prompt, stringProperty } from "@oh-my-pi/pi-utils"; +import { reset as resetCapabilities } from "../capability"; +import type { ModelRegistry } from "../config/model-registry"; +import { formatModelString } from "../config/model-resolver"; +import type { Settings, SkillsSettings } from "../config/settings"; +import type { CustomTool, CustomToolContext } from "../extensibility/custom-tools/types"; +import { CustomToolAdapter } from "../extensibility/custom-tools/wrapper"; +import type { ExtensionRunner } from "../extensibility/extensions"; +import { ExtensionToolWrapper } from "../extensibility/extensions/wrapper"; +import { loadSkills, type Skill, type SkillWarning, setActiveSkills } from "../extensibility/skills"; +import type { LocalProtocolOptions } from "../internal-urls"; +import { resolveMemoryBackend } from "../memory-backend/resolve"; +import { MEMORY_BACKEND_TOOL_NAMES } from "../memory-backend/tool-names"; +import type { MemoryBackendStartOptions } from "../memory-backend/types"; +import xdevMountNoticePrompt from "../prompts/system/xdev-mount-notice.md" with { type: "text" }; +import { usesCodexTaskPrompt } from "../task/prompt-policy"; +import { isMCPToolName, normalizeToolNames } from "../tools/builtin-names"; +import { wrapToolWithMetaNotice } from "../tools/output-meta"; +import { ToolAbortError, ToolError } from "../tools/tool-errors"; +import { isMountableUnderXdev, type XdevRegistry } from "../tools/xdev"; +import { type EditMode, resolveEditMode } from "../utils/edit-mode"; +import { formatLocalCalendarDate } from "../utils/local-date"; +import { + extractPermissionLocations, + getPermissionIntent, + PERMISSION_OPTIONS, + PERMISSION_OPTIONS_BY_ID, + PERMISSION_REQUIRED_TOOLS, +} from "./acp-permission-gate"; +import type { ClientBridge, ClientBridgePermissionOutcome } from "./client-bridge"; +import type { CustomMessage } from "./messages"; +import type { SessionManager } from "./session-manager"; + +/** Capabilities borrowed from the owning AgentSession. */ +export interface SessionToolsHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + modelRegistry: ModelRegistry; + extensionRunner(): ExtensionRunner | undefined; + clientBridge(): ClientBridge | undefined; + agentKind(): "main" | "sub"; + isDisposed(): boolean; + isStreaming(): boolean; + queuedMessageCount(): number; + planModeEnabled(): boolean; + model(): Model | undefined; + memoryBackendSession(): MemoryBackendStartOptions["session"]; + clearInheritedProviderPromptCacheKey(): void; + clearMemoryPromotionSnapshot(): void; + captureMemoryPromotionSnapshot(prompt: string[]): void; + emitNotice(level: "info" | "warning" | "error", message: string, source?: string): void; + notifyCommandMetadataChanged(): void; + localProtocolOptions(): LocalProtocolOptions; +} + +interface SessionToolsOptions { + autoApprove?: boolean; + toolRegistry?: Map; + createVibeTools?: () => AgentTool[]; + builtInToolNames?: Iterable; + presentationPinnedToolNames?: ReadonlySet; + ensureWriteRegistered?: () => Promise; + rebuildSystemPrompt?: (toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>; + getLocalCalendarDate?: () => string; + getMcpServerInstructions?: () => Map | undefined; + xdevRegistry?: XdevRegistry; + initialMountedXdevToolNames?: string[]; + setActiveToolNames?: (names: Iterable) => void; + baseSystemPrompt: string[]; + skills?: Skill[]; + skillWarnings?: SkillWarning[]; + skillsSettings?: SkillsSettings; + skillsReloadable?: boolean; +} + +const XDEV_MOUNT_NOTICE_MESSAGE_TYPE = "xdev-mount-notice"; + +/** Owns tool registration, presentation, prompt rebuilding, skills, and permissions. */ +export class SessionTools { + readonly #host: SessionToolsHost; + #autoApprove: boolean; + #toolRegistry: Map; + #createVibeTools: (() => AgentTool[]) | undefined; + #installedVibeToolNames = new Set(); + #builtInToolNames: Set; + #rpcHostToolNames = new Set(); + #xdevRegistry: XdevRegistry | undefined; + #mountedXdevToolNames: Set; + #pendingXdevMountDelta: { added: Set; removed: Set } | undefined; + #presentationPinnedToolNames: ReadonlySet | undefined; + #runtimeSelectedToolNames: ReadonlySet | undefined; + #baseSystemPrompt: string[]; + #lastAppliedToolSignature: string | undefined; + #promptModelKey: string | undefined; + #rebuildSystemPrompt: SessionToolsOptions["rebuildSystemPrompt"]; + #getLocalCalendarDate: () => string; + #getMcpServerInstructions: SessionToolsOptions["getMcpServerInstructions"]; + #setActiveToolNames: SessionToolsOptions["setActiveToolNames"]; + #ensureWriteRegistered: SessionToolsOptions["ensureWriteRegistered"]; + #skills: Skill[]; + #skillWarnings: SkillWarning[]; + #skillsSettings: SkillsSettings | undefined; + #skillsReloadable: boolean; + #acpPermissionDecisions = new Map(); + + constructor(host: SessionToolsHost, options: SessionToolsOptions) { + this.#host = host; + this.#autoApprove = options.autoApprove === true; + this.#toolRegistry = options.toolRegistry ?? new Map(); + this.#createVibeTools = options.createVibeTools; + this.#builtInToolNames = new Set(options.builtInToolNames ?? []); + this.#presentationPinnedToolNames = options.presentationPinnedToolNames; + this.#ensureWriteRegistered = options.ensureWriteRegistered; + this.#rebuildSystemPrompt = options.rebuildSystemPrompt; + this.#getLocalCalendarDate = options.getLocalCalendarDate ?? formatLocalCalendarDate; + this.#getMcpServerInstructions = options.getMcpServerInstructions; + this.#xdevRegistry = options.xdevRegistry; + this.#mountedXdevToolNames = new Set(options.initialMountedXdevToolNames ?? []); + this.#setActiveToolNames = options.setActiveToolNames; + this.#baseSystemPrompt = options.baseSystemPrompt; + this.#skills = options.skills ?? []; + this.#skillWarnings = options.skillWarnings ?? []; + this.#skillsSettings = options.skillsSettings; + this.#skillsReloadable = options.skillsReloadable ?? true; + this.#promptModelKey = this.#currentPromptModelKey(); + } + + /** Mutable registry shared with controller hosts that inspect available tools. */ + get registry(): Map { + return this.#toolRegistry; + } + + /** Current stable base system prompt. */ + get baseSystemPrompt(): string[] { + return this.#baseSystemPrompt; + } + + /** Replaces the controller-owned base prompt without applying it to the agent. */ + setBaseSystemPrompt(prompt: string[]): void { + this.#baseSystemPrompt = prompt; + } + + /** Skills currently rendered into the system prompt. */ + get skills(): Skill[] { + return this.#skills; + } + + /** Diagnostics produced while loading the current skills. */ + get skillWarnings(): SkillWarning[] { + return this.#skillWarnings; + } + + /** Settings snapshot used for the current skill discovery. */ + get skillsSettings(): SkillsSettings | undefined { + return this.#skillsSettings; + } + + /** Re-wraps active and mounted tools after the ACP client changes. */ + refreshAcpPermissionGates(): void { + this.#acpPermissionDecisions.clear(); + const activeTools = this.getActiveToolNames() + .map(name => this.#toolRegistry.get(name)) + .filter((tool): tool is AgentTool => tool !== undefined) + .map(tool => this.#wrapToolForAcpPermission(tool)); + this.#host.agent.setTools(activeTools); + const mountedTools = [...this.#mountedXdevToolNames] + .map(name => this.#toolRegistry.get(name)) + .filter((tool): tool is AgentTool => tool !== undefined) + .map(tool => this.#wrapToolForAcpPermission(tool)); + this.#xdevRegistry?.reconcile(mountedTools); + } + + #getActiveNonMCPToolNames(): string[] { + return this.getEnabledToolNames().filter(name => !isMCPToolName(name) && this.#toolRegistry.has(name)); + } + + /** Names of tools currently exposed at the top level. */ + getActiveToolNames(): string[] { + return this.#host.agent.state.tools.map(t => t.name); + } + + /** Enabled top-level and discoverable tool names. */ + getEnabledToolNames(): string[] { + if (this.#mountedXdevToolNames.size === 0) return this.getActiveToolNames(); + return [...this.getActiveToolNames(), ...this.#mountedXdevToolNames]; + } + + /** Names of dynamic tools mounted under `xd://`. */ + getMountedXdevToolNames(): string[] { + return [...this.#mountedXdevToolNames]; + } + + /** Whether the edit tool is registered. */ + get hasEditTool(): boolean { + return this.#toolRegistry.has("edit"); + } + + /** Looks up a registered tool by name. */ + getToolByName(name: string): AgentTool | undefined { + return this.#toolRegistry.get(name); + } + + /** Whether a registry entry came from a built-in factory. */ + hasBuiltInTool(name: string): boolean { + return this.#builtInToolNames.has(name); + } + + /** Names of every registered tool. */ + getAllToolNames(): string[] { + return Array.from(this.#toolRegistry.keys()); + } + + #wrapRuntimeTool(tool: AgentTool): AgentTool { + const wrapped = wrapToolWithMetaNotice(tool); + const extensionRunner = this.#host.extensionRunner(); + return extensionRunner ? new ExtensionToolWrapper(wrapped, extensionRunner) : wrapped; + } + + /** Installs and activates the ephemeral vibe tool set. */ + async activateVibeTools(baseToolNames: string[]): Promise { + const createVibeTools = this.#createVibeTools; + if (!createVibeTools) { + throw new Error("Vibe tools are unavailable in this session."); + } + + const tools = createVibeTools(); + const vibeToolNames = tools.map(tool => tool.name); + if (new Set(vibeToolNames).size !== vibeToolNames.length) { + throw new Error("Vibe tool names must be unique."); + } + + for (const tool of tools) { + if (this.#toolRegistry.has(tool.name)) continue; + this.#toolRegistry.set(tool.name, this.#wrapRuntimeTool(tool)); + this.#builtInToolNames.add(tool.name); + this.#installedVibeToolNames.add(tool.name); + } + + await this.applyActiveToolsByName([...new Set([...baseToolNames, ...vibeToolNames])]); + } + + /** Uninstalls vibe tools and activates the replacement set. */ + async deactivateVibeTools(nextToolNames: string[]): Promise { + this.#uninstallVibeTools(); + await this.applyActiveToolsByName(nextToolNames); + } + + /** Removes vibe tools without restoring a source-session snapshot. */ + async removeVibeToolsPreservingActive(): Promise { + const removed = new Set(this.#installedVibeToolNames); + this.#uninstallVibeTools(); + const nextActive = this.getActiveToolNames().filter(name => !removed.has(name)); + await this.applyActiveToolsByName(nextActive); + } + + #uninstallVibeTools(): void { + for (const name of this.#installedVibeToolNames) { + this.#toolRegistry.delete(name); + this.#builtInToolNames.delete(name); + } + this.#installedVibeToolNames.clear(); + } + + #getEditModeSession() { + return { + settings: this.#host.settings, + getActiveModelString: () => { + const model = this.#host.model(); + return model ? formatModelString(model) : undefined; + }, + } as const; + } + + /** Resolves the edit mode for the active model and settings. */ + resolveActiveEditMode(): EditMode { + return resolveEditMode(this.#getEditModeSession()); + } + + #currentPromptModelKey(): string | undefined { + const activeModel = this.#host.model(); + const model = activeModel ? formatModelString(activeModel) : undefined; + if (!model || this.#host.settings.get("includeModelInPrompt")) return model; + return usesCodexTaskPrompt(model) ? "task-policy:gpt-5.6" : "task-policy:default"; + } + + /** Rebuilds model-dependent tool prompts after a model change. */ + async syncAfterModelChange(previousEditMode: EditMode): Promise { + const currentEditMode = this.resolveActiveEditMode(); + const editModeChanged = previousEditMode !== currentEditMode && this.getActiveToolNames().includes("edit"); + // The system prompt selects model-specific policy even when it does not display the model id. + const modelChanged = this.#currentPromptModelKey() !== this.#promptModelKey; + if (editModeChanged || modelChanged) { + await this.refreshBaseSystemPrompt(); + } + } + + /** Enabled MCP tools in their current presentation partition. */ + getSelectedMCPToolNames(): string[] { + // Every connected MCP tool is enabled; presentation (top-level vs xd://) is + // decided by loadMode. Return the enabled MCP tools in the current set. + return this.getEnabledToolNames().filter(name => isMCPToolName(name) && this.#toolRegistry.has(name)); + } + + /** + * Wrap a tool with a permission-gate proxy when an ACP client is connected. + * Only wraps tools whose name is in PERMISSION_REQUIRED_TOOLS and only when + * the bridge exposes `requestPermission`. No-ops for all other cases. + * + * When the user has explicitly opted into `yolo` / auto-approve behavior (via + * the SDK/CLI `autoApprove` flag or a configured `tools.approvalMode: yolo`), + * skips the gate unless the per-tool policy explicitly requires a prompt or + * deny. The schema default is also `yolo`, so an explicit configuration or + * explicit session flag is required: default-config ACP sessions keep the + * client-side permission gate. + */ + #wrapToolForAcpPermission(tool: T): T { + const bridge = this.#host.clientBridge(); + // Match the capability+method gating pattern used by read/write/bash. + if (!bridge?.capabilities.requestPermission || !bridge.requestPermission) return tool; + if (PERMISSION_REQUIRED_TOOLS[tool.name] !== true) return tool; + // Skip the gate only on explicit yolo opt-in; honour per-tool policies + // that require a prompt or deny (matching the normal approval wrapper). + if (this.#isExplicitAutoApproveMode()) { + const userPolicies = (this.#host.settings.get("tools.approval") ?? {}) as Record; + const toolPolicy = userPolicies[tool.name]; + if (!toolPolicy || toolPolicy === "allow") return tool; + } + return new Proxy(tool, { + get: (target, prop) => { + if (prop !== "execute") return target[prop as keyof T]; + return async ( + toolCallId: string, + args: unknown, + signal: AbortSignal | undefined, + onUpdate: never, + ctx: never, + ) => { + const permissionIntent = getPermissionIntent(target.name, args); + if (!permissionIntent) { + return await target.execute(toolCallId, args as never, signal, onUpdate, ctx); + } + const command = + target.name === "bash" && args && typeof args === "object" && !Array.isArray(args) + ? stringProperty(args, "command") + : undefined; + const commandContent = command + ? [{ type: "content" as const, content: { type: "text" as const, text: `$ ${command}` } }] + : undefined; + // Short-circuit on persisted decisions. + const persisted = this.#acpPermissionDecisions.get(permissionIntent.cacheKey); + if (persisted === "allow_always") { + return await target.execute(toolCallId, args as never, signal, onUpdate, ctx); + } + if (persisted === "reject_always") { + throw new ToolError(`Tool call rejected by user (preference)`); + } + if (signal?.aborted) { + throw new ToolAbortError("Permission request cancelled"); + } + type PermissionRaceResult = + | { kind: "permission"; outcome: ClientBridgePermissionOutcome } + | { kind: "aborted" }; + const { promise: abortPromise, resolve: resolveAbort } = Promise.withResolvers(); + const onAbort = () => resolveAbort({ kind: "aborted" }); + signal?.addEventListener("abort", onAbort, { once: true }); + let raced: PermissionRaceResult; + try { + const permissionPromise = bridge.requestPermission!( + { + toolCallId, + toolName: target.name, + title: permissionIntent.title, + ...(target.name === "bash" ? { kind: "execute" } : {}), + status: "pending", + rawInput: args, + ...(commandContent ? { content: commandContent } : {}), + locations: extractPermissionLocations( + args, + this.#host.sessionManager.getCwd(), + permissionIntent.paths, + ), + }, + PERMISSION_OPTIONS, + signal, + ).then(outcome => ({ kind: "permission" as const, outcome })); + raced = await Promise.race([permissionPromise, abortPromise]); + } finally { + signal?.removeEventListener("abort", onAbort); + } + if (raced.kind === "aborted" || signal?.aborted) { + throw new ToolAbortError("Permission request cancelled"); + } + const outcome = raced.outcome; + if (outcome.outcome === "cancelled") { + throw new ToolAbortError("Permission request cancelled"); + } + const selectedOption = PERMISSION_OPTIONS_BY_ID[outcome.optionId]; + if (!selectedOption) { + throw new ToolError(`Tool permission response used unknown option ID: ${outcome.optionId}`); + } + if (selectedOption.kind === "allow_always") { + this.#acpPermissionDecisions.set(permissionIntent.cacheKey, "allow_always"); + } else if (selectedOption.kind === "reject_always") { + this.#acpPermissionDecisions.set(permissionIntent.cacheKey, "reject_always"); + } + if (selectedOption.kind === "reject_once" || selectedOption.kind === "reject_always") { + throw new ToolError(`Tool call rejected by user (${target.name})`); + } + return await target.execute(toolCallId, args as never, signal, onUpdate, ctx); + }; + }, + }) as T; + } + + #isExplicitAutoApproveMode(): boolean { + return ( + this.#autoApprove || + (this.#host.settings.isConfigured("tools.approvalMode") && + this.#host.settings.get("tools.approvalMode") === "yolo") + ); + } + + /** Applies an enabled tool set and reconciles its `xd://` partition. */ + async applyActiveToolsByName(toolNames: string[]): Promise { + toolNames = normalizeToolNames(toolNames); + const selectedTools = toolNames.flatMap(name => { + const tool = this.#toolRegistry.get(name); + return tool ? [{ name, tool }] : []; + }); + const xdevReadAvailable = this.#builtInToolNames.has("read") && selectedTools.some(({ name }) => name === "read"); + const isPresentationPinned = (name: string): boolean => + this.#presentationPinnedToolNames?.has(name) === true || this.#runtimeSelectedToolNames?.has(name) === true; + const mountCandidates = selectedTools.filter( + ({ name, tool }) => + this.#xdevRegistry !== undefined && + xdevReadAvailable && + !isPresentationPinned(name) && + isMountableUnderXdev(tool), + ); + + let builtInWriteAvailable = this.#builtInToolNames.has("write"); + if (mountCandidates.length > 0 && !builtInWriteAvailable) { + builtInWriteAvailable = (await this.#ensureWriteRegistered?.()) === true; + if (builtInWriteAvailable) this.#builtInToolNames.add("write"); + } + const mountNames = builtInWriteAvailable ? new Set(mountCandidates.map(({ name }) => name)) : new Set(); + const tools: AgentTool[] = []; + const validToolNames: string[] = []; + const mountedTools: AgentTool[] = []; + for (const { name, tool } of selectedTools) { + if (mountNames.has(name)) { + mountedTools.push(this.#wrapToolForAcpPermission(tool)); + } else { + tools.push(this.#wrapToolForAcpPermission(tool)); + validToolNames.push(name); + } + } + + const pinnedWrite = isPresentationPinned("write"); + const activeDeferrableTool = tools.some(tool => tool.deferrable === true); + const transportNeeded = mountedTools.length > 0 || activeDeferrableTool || this.#host.planModeEnabled(); + if (transportNeeded && !builtInWriteAvailable) { + builtInWriteAvailable = (await this.#ensureWriteRegistered?.()) === true; + if (builtInWriteAvailable) this.#builtInToolNames.add("write"); + } + if (transportNeeded && builtInWriteAvailable) { + const write = this.#toolRegistry.get("write"); + if (write && !validToolNames.includes("write")) { + tools.push(this.#wrapToolForAcpPermission(write)); + validToolNames.push("write"); + } + } else if ( + !pinnedWrite && + (this.#presentationPinnedToolNames !== undefined || this.#runtimeSelectedToolNames !== undefined) + ) { + const writeNameIndex = validToolNames.indexOf("write"); + if (writeNameIndex >= 0 && this.#builtInToolNames.has("write")) validToolNames.splice(writeNameIndex, 1); + const writeToolIndex = tools.findIndex(tool => tool.name === "write" && this.#builtInToolNames.has("write")); + if (writeToolIndex >= 0) tools.splice(writeToolIndex, 1); + } + + const previousMounted = this.#mountedXdevToolNames; + const previousMountedTools = [...previousMounted].flatMap(name => { + const tool = this.#xdevRegistry?.get(name); + return tool ? [tool] : []; + }); + const previousActiveToolNames = this.getActiveToolNames(); + this.#mountedXdevToolNames = new Set(mountedTools.map(tool => tool.name)); + this.#xdevRegistry?.reconcile(mountedTools); + this.#setActiveToolNames?.(validToolNames); + + let rebuiltSystemPrompt: string[] | undefined; + let rebuiltSignature: string | undefined; + try { + if (this.#rebuildSystemPrompt) { + const signature = this.#computeAppliedToolSignature(validToolNames, tools); + if (signature !== this.#lastAppliedToolSignature) { + const built = await this.#rebuildSystemPrompt(validToolNames, this.#toolRegistry); + rebuiltSystemPrompt = built.systemPrompt; + rebuiltSignature = signature; + } + } + } catch (error) { + this.#mountedXdevToolNames = previousMounted; + this.#xdevRegistry?.reconcile(previousMountedTools); + this.#setActiveToolNames?.(previousActiveToolNames); + throw error; + } + + this.#notifyXdevMountDelta(previousMounted); + this.#host.agent.setTools(tools); + if (rebuiltSystemPrompt && rebuiltSignature) { + if (this.#lastAppliedToolSignature !== undefined) this.#host.clearInheritedProviderPromptCacheKey(); + this.#baseSystemPrompt = rebuiltSystemPrompt; + this.#host.clearMemoryPromotionSnapshot(); + this.#host.agent.setSystemPrompt(this.#baseSystemPrompt); + this.#lastAppliedToolSignature = rebuiltSignature; + this.#promptModelKey = this.#currentPromptModelKey(); + } + } + + /** + * Record a mid-session `xd://` mount delta for the model without rewriting + * the system prompt: the prompt (and its provider cache prefix) stays + * byte-stable across MCP connects and disconnects. The delta is NOT steered + * immediately — a steered notice landing at a run's stop boundary (or while + * the session is idle) forces an unsolicited extra assistant turn — it is + * coalesced into {@link #pendingXdevMountDelta} and rides along with the + * next prompt (docs + schema stay one `read xd://` away). The full + * docs join the system prompt opportunistically on the next unrelated + * rebuild. + */ + #notifyXdevMountDelta(previousMounted: ReadonlySet): void { + const registry = this.#xdevRegistry; + if (!registry) return; + const current = this.#mountedXdevToolNames; + const addedNames = [...current].filter(name => !previousMounted.has(name)); + const removedNames = [...previousMounted].filter(name => !current.has(name)); + if (addedNames.length === 0 && removedNames.length === 0) return; + // Coalesce against the unannounced delta: an unmount cancels a pending + // mount the model never learned about, and a remount cancels a pending + // unmount. + const pending = this.#pendingXdevMountDelta ?? { added: new Set(), removed: new Set() }; + for (const name of addedNames) { + if (!pending.removed.delete(name)) pending.added.add(name); + } + for (const name of removedNames) { + if (!pending.added.delete(name)) pending.removed.add(name); + } + this.#pendingXdevMountDelta = pending.added.size > 0 || pending.removed.size > 0 ? pending : undefined; + if (this.#host.settings.get("startup.quiet")) return; + const parts: string[] = []; + if (addedNames.length > 0) parts.push(`mounted ${addedNames.join(", ")}`); + if (removedNames.length > 0) parts.push(`unmounted ${removedNames.join(", ")}`); + this.#host.emitNotice("info", `xd://: ${parts.join("; ")}`, "xdev"); + } + + /** Consumes the hidden notice for unannounced `xd://` mount changes. */ + takePendingXdevMountNotice(): CustomMessage | undefined { + const pending = this.#pendingXdevMountDelta; + if (!pending) return undefined; + this.#pendingXdevMountDelta = undefined; + const summaries = new Map(this.#xdevRegistry?.entries().map(entry => [entry.name, entry.summary]) ?? []); + const added = [...pending.added].map(name => ({ name, summary: summaries.get(name) ?? "" })); + const removed = [...pending.removed].map(name => ({ name })); + return { + role: "custom", + customType: XDEV_MOUNT_NOTICE_MESSAGE_TYPE, + content: prompt.render(xdevMountNoticePrompt, { added, removed }), + attribution: "agent", + display: false, + timestamp: Date.now(), + }; + } + + /** Rediscovers reloadable skills and refreshes prompt metadata. */ + async refreshSkills(): Promise { + if (!this.#skillsReloadable) { + return; + } + + resetCapabilities(); + const skillsSettings = this.#host.settings.getGroup("skills"); + const discovered = await loadSkills({ + ...skillsSettings, + cwd: this.#host.sessionManager.getCwd(), + disabledExtensions: this.#host.settings.get("disabledExtensions") ?? [], + }); + this.#skills = discovered.skills; + this.#skillWarnings = discovered.warnings; + this.#skillsSettings = skillsSettings; + + if (this.#host.agentKind() === "main") { + setActiveSkills(this.#skills); + } + await this.refreshBaseSystemPrompt(); + this.#host.notifyCommandMetadataChanged(); + } + + /** Selects enabled tools, ignoring names absent from the registry. */ + async setActiveToolsByName(toolNames: string[]): Promise { + const normalized = normalizeToolNames(toolNames); + // Transport-write eligibility keys off the *current* active set: an ordinary + // selection change should not demote `write` unless it is already active. + await this.#applyToolPresentation( + normalized, + this.#mountedXdevToolNames, + this.getActiveToolNames().includes("write"), + ); + } + + /** + * Restore an enabled tool set with its exact top-level versus `xd://` partition. + * + * Both inputs are required because {@link setActiveToolsByName} only receives the + * enabled name list and classifies mounts from the current `#mountedXdevToolNames`. + * Rollback/restore callers must pass the snapshotted mounted subset so names that + * were top-level stay pinned (`#runtimeSelectedToolNames`) and names that were under + * `xd://` remain mount-eligible, even when the live mount set has drifted. + * + * Names outside `mountedToolNames` are pinned top-level for this application; + * names in the mounted subset remain eligible for xdev mounting. Delegates the + * actual apply through {@link applyActiveToolsByName} and restores the prior runtime + * selection if that apply throws. + */ + async setActiveToolPresentation(toolNames: string[], mountedToolNames: string[]): Promise { + const normalized = normalizeToolNames(toolNames); + // Restoration targets a snapshot, so write eligibility comes from the + // *target* set rather than whatever happens to be active mid-rollback. + await this.#applyToolPresentation( + normalized, + new Set(normalizeToolNames(mountedToolNames)), + normalized.includes("write"), + ); + } + + /** + * Shared body for {@link setActiveToolsByName} and {@link setActiveToolPresentation}: + * pins non-mounted names as the runtime selection (holding `write` back when it is + * transport-only) and applies the set, rolling the selection back if apply throws. + */ + async #applyToolPresentation( + normalized: string[], + mounted: ReadonlySet, + writeSelected: boolean, + ): Promise { + const transportWriteActive = + writeSelected && + this.#builtInToolNames.has("write") && + this.#presentationPinnedToolNames?.has("write") !== true && + this.#runtimeSelectedToolNames?.has("write") !== true && + (mounted.size > 0 || this.#host.planModeEnabled()); + const previousRuntimeSelectedToolNames = this.#runtimeSelectedToolNames; + this.#runtimeSelectedToolNames = new Set( + normalized.filter(name => !mounted.has(name) && !(name === "write" && transportWriteActive)), + ); + try { + await this.applyActiveToolsByName(normalized); + } catch (error) { + this.#runtimeSelectedToolNames = previousRuntimeSelectedToolNames; + throw error; + } + } + + /** Replaces memory-backend tools while preserving unrelated selections. */ + async replaceMemoryTools(tools: AgentTool[]): Promise { + const removed = new Set(MEMORY_BACKEND_TOOL_NAMES.filter(name => this.#builtInToolNames.has(name))); + const nextActive = this.getEnabledToolNames().filter(name => !removed.has(name)); + for (const name of removed) { + this.#toolRegistry.delete(name); + this.#builtInToolNames.delete(name); + } + + for (const tool of tools) { + if (!MEMORY_BACKEND_TOOL_NAMES.some(name => name === tool.name) || this.#toolRegistry.has(tool.name)) { + continue; + } + const wrapped = this.#wrapRuntimeTool(tool); + this.#toolRegistry.set(wrapped.name, wrapped); + this.#builtInToolNames.add(wrapped.name); + nextActive.push(wrapped.name); + } + await this.applyActiveToolsByName([...new Set(nextActive)]); + } + + /** Rebuilds the stable base prompt for the current tools and model. */ + async refreshBaseSystemPrompt(): Promise { + if (this.#host.isDisposed() || !this.#rebuildSystemPrompt) return; + const activeToolNames = this.getActiveToolNames(); + this.#setActiveToolNames?.(activeToolNames); + const previousBaseSystemPrompt = this.#baseSystemPrompt; + const built = await this.#rebuildSystemPrompt(activeToolNames, this.#toolRegistry); + if (this.#host.isDisposed()) return; + this.#baseSystemPrompt = built.systemPrompt; + this.#host.clearMemoryPromotionSnapshot(); + if ( + previousBaseSystemPrompt.length !== this.#baseSystemPrompt.length || + previousBaseSystemPrompt.some((part, index) => part !== this.#baseSystemPrompt[index]) + ) { + this.#host.clearInheritedProviderPromptCacheKey(); + } + this.#host.agent.setSystemPrompt(this.#baseSystemPrompt); + this.#promptModelKey = this.#currentPromptModelKey(); + // Refresh the cached signature so a subsequent `applyActiveToolsByName` with + // the same tool set does not re-rebuild on top of the explicit refresh we + // just performed (and conversely, a different set forces a fresh rebuild). + const activeTools = activeToolNames + .map(name => this.#toolRegistry.get(name)) + .filter((tool): tool is AgentTool => tool != null); + this.#lastAppliedToolSignature = this.#computeAppliedToolSignature(activeToolNames, activeTools); + } + + /** Applies one-turn memory prompt injection before an agent run. */ + async buildSystemPromptForAgentStart(promptText: string): Promise { + const backend = await resolveMemoryBackend(this.#host.settings); + if (!backend.beforeAgentStartPrompt) return this.#baseSystemPrompt; + + try { + const injected = await backend.beforeAgentStartPrompt(this.#host.memoryBackendSession(), promptText); + if (!injected) return this.#baseSystemPrompt; + + const previousBaseSystemPrompt = this.#baseSystemPrompt; + try { + await this.refreshBaseSystemPrompt(); + } catch (refreshErr) { + logger.debug("Memory backend prompt refresh after beforeAgentStartPrompt failed", { + backend: backend.id, + error: String(refreshErr), + }); + } + + if ( + this.#baseSystemPrompt.length !== previousBaseSystemPrompt.length || + this.#baseSystemPrompt.some((part, index) => part !== previousBaseSystemPrompt[index]) + ) { + return this.#baseSystemPrompt; + } + + this.#host.captureMemoryPromotionSnapshot(previousBaseSystemPrompt); + const stablePrompt = [...previousBaseSystemPrompt, injected]; + this.#baseSystemPrompt = stablePrompt; + this.#host.agent.setSystemPrompt(stablePrompt); + return stablePrompt; + } catch (err) { + logger.debug("Memory backend beforeAgentStartPrompt failed", { + backend: backend.id, + error: String(err), + }); + return this.#baseSystemPrompt; + } + } + + /** + * Compose a stable signature for the inputs that `rebuildSystemPrompt` reads. + * Two calls producing identical signatures are guaranteed to produce identical + * system prompt bytes, so the rebuild can be skipped. + * + * The signature covers: + * 1. Active tool names in order (the prompt renders them in this order). + * 2. Active tool labels, descriptions, and wire-visible names — all are + * rendered into the prompt body (see `system-prompt.md` `{{label}}: \`{{name}}\`` + * and `toolPromptNames` in `buildSystemPrompt`). The wire name comes from + * `tool.customWireName` and overrides the internal name on the model wire + * (e.g. `edit` exposes itself as `apply_patch` to GPT-5 in apply_patch mode); + * a stale wire name would desync prompt guidance from actual tool routing. + * 3. When MCP discovery is on, every registry tool's name+label+description+ + * customWireName, since `rebuildSystemPrompt` summarizes discoverable MCP + * tools that are not in the active set. + * 4. MCP server instructions text (per server), since `rebuildSystemPrompt` + * embeds these in the appended prompt under "## MCP Server Instructions". + * A server upgrade can change instructions while keeping tools identical. + * + * Settings-driven tool metadata is covered automatically: built-in tools that + * depend on settings expose `description`/`label` via getters (see `TaskTool`, + * `SearchToolBm25Tool`, `EditTool`), and the signature reads them live on every + * call - so a settings flip that mutates the rendered string differs the signature + * the next time {@link applyActiveToolsByName} runs. Do not refactor `describeTool` + * to cache per-tool strings without preserving this property. + * + * Inputs NOT covered: tool input schemas; memory instructions read from disk; + * and SDK-init-time closure constants in `sdk.ts` (`inlineToolDescriptors`, + * `eagerTasks`, `intentField`, `mcpDiscoveryEnabled`, `secretsEnabled`). The + * closure-captured ones cannot change at runtime regardless of skip behavior. + * For everything else, callers must explicitly call {@link refreshBaseSystemPrompt} + * after side-effecting changes; see the memory hooks and {@link syncAfterModelChange}. + * + * The current calendar date IS covered (appended as a segment) because + * `buildSystemPrompt` injects it into the prompt body (`Today is '{{date}}'`). + * Without this, a session spanning midnight with only tool-stable MCP + * reconnects would keep yesterday's date indefinitely. + */ + #computeAppliedToolSignature(toolNames: string[], tools: AgentTool[]): string { + // Order-preserving join: any reorder must produce a different signature so + // the rebuild fires and the new tool list reaches the API. + const nameSegment = toolNames.join("\u0001"); + const describeTool = (tool: AgentTool): string => + `${tool.name}=${tool.label ?? ""}|${tool.description ?? ""}|${tool.customWireName ?? ""}`; + const descriptionSegment = tools.map(describeTool).join("\u0002"); + let instructionsSegment = ""; + const serverInstructions = this.#getMcpServerInstructions?.(); + if (serverInstructions && serverInstructions.size > 0) { + // Sort by server name so transport flap order does not perturb the signature. + const entries: string[] = []; + for (const [server, instructions] of serverInstructions) { + entries.push(`${server}=${instructions}`); + } + entries.sort(); + instructionsSegment = entries.join("\u0006"); + } + // The xd:// device inventory is deliberately NOT part of the signature: + // a mount/unmount announces itself via `#notifyXdevMountDelta` instead of + // rewriting the system prompt, so MCP connects/disconnects keep the + // prompt (and its provider cache prefix) byte-stable. Rebuilds triggered + // by other inputs pick up the current device docs opportunistically. + const date = this.#getLocalCalendarDate(); + return `${nameSegment}\u0003${descriptionSegment}\u0007${instructionsSegment}|${date}`; + } + + /** + * Replace MCP tools in the registry and enable them immediately. Every + * connected MCP tool becomes available (mounted under `xd://` when that + * transport is active, else top-level). Lets `/mcp add/remove/reauth` take + * effect without restarting the session. + */ + async refreshMCPTools(mcpTools: CustomTool[]): Promise { + const existingNames = Array.from(this.#toolRegistry.keys()); + const previousMcpTools = new Map( + existingNames.flatMap(name => { + const tool = this.#toolRegistry.get(name); + return isMCPToolName(name) && tool ? [[name, tool] as const] : []; + }), + ); + for (const name of existingNames) { + if (isMCPToolName(name)) { + this.#toolRegistry.delete(name); + } + } + + const getCustomToolContext = (): CustomToolContext => ({ + sessionManager: this.#host.sessionManager, + modelRegistry: this.#host.modelRegistry, + model: this.#host.model(), + isIdle: () => !this.#host.isStreaming(), + hasQueuedMessages: () => this.#host.queuedMessageCount() > 0, + abort: () => { + this.#host.agent.abort(); + }, + settings: this.#host.settings, + localProtocolOptions: this.#host.localProtocolOptions(), + }); + + const extensionRunner = this.#host.extensionRunner(); + for (const customTool of mcpTools) { + const wrapped = wrapToolWithMetaNotice(CustomToolAdapter.wrap(customTool, getCustomToolContext) as AgentTool); + const finalTool = ( + extensionRunner ? new ExtensionToolWrapper(wrapped, extensionRunner) : wrapped + ) as AgentTool; + this.#toolRegistry.set(finalTool.name, finalTool); + } + + // Every connected MCP tool is selected; centralized repartitioning owns + // presentation pins and write-transport activation/removal. + const nextActive = [...new Set([...this.#getActiveNonMCPToolNames(), ...mcpTools.map(tool => tool.name)])]; + try { + await this.applyActiveToolsByName(nextActive); + } catch (error) { + for (const name of this.#toolRegistry.keys()) { + if (isMCPToolName(name)) this.#toolRegistry.delete(name); + } + for (const [name, tool] of previousMcpTools) this.#toolRegistry.set(name, tool); + throw error; + } + } + + /** Replaces RPC host-owned tools and refreshes the active set before the next model call. */ + async refreshRpcHostTools(rpcTools: AgentTool[]): Promise { + const nextToolNames = rpcTools.map(tool => tool.name); + const uniqueToolNames = new Set(nextToolNames); + if (uniqueToolNames.size !== nextToolNames.length) { + throw new Error("RPC host tool names must be unique"); + } + + for (const name of uniqueToolNames) { + if (this.#toolRegistry.has(name) && !this.#rpcHostToolNames.has(name)) { + throw new Error(`RPC host tool "${name}" conflicts with an existing tool`); + } + } + + const previousRpcHostToolNames = new Set(this.#rpcHostToolNames); + const previousActiveToolNames = this.getEnabledToolNames(); + const previousRpcHostTools = new Map( + [...previousRpcHostToolNames].flatMap(name => { + const tool = this.#toolRegistry.get(name); + return tool ? [[name, tool] as const] : []; + }), + ); + for (const name of previousRpcHostToolNames) { + this.#toolRegistry.delete(name); + } + this.#rpcHostToolNames.clear(); + + const extensionRunner = this.#host.extensionRunner(); + for (const tool of rpcTools) { + const metaWrapped = wrapToolWithMetaNotice(tool); + const finalTool = ( + extensionRunner ? new ExtensionToolWrapper(metaWrapped, extensionRunner) : metaWrapped + ) as AgentTool; + this.#toolRegistry.set(finalTool.name, finalTool); + this.#rpcHostToolNames.add(finalTool.name); + } + + const activeNonRpcToolNames = previousActiveToolNames.filter(name => !previousRpcHostToolNames.has(name)); + const preservedRpcToolNames = previousActiveToolNames.filter( + name => previousRpcHostToolNames.has(name) && this.#rpcHostToolNames.has(name), + ); + const autoActivatedRpcToolNames = rpcTools + .filter(tool => !tool.hidden && !previousRpcHostToolNames.has(tool.name)) + .map(tool => tool.name); + try { + await this.applyActiveToolsByName( + Array.from(new Set([...activeNonRpcToolNames, ...preservedRpcToolNames, ...autoActivatedRpcToolNames])), + ); + } catch (error) { + for (const name of this.#rpcHostToolNames) this.#toolRegistry.delete(name); + this.#rpcHostToolNames = previousRpcHostToolNames; + for (const [name, tool] of previousRpcHostTools) this.#toolRegistry.set(name, tool); + throw error; + } + } +} diff --git a/packages/coding-agent/src/session/stream-guards.ts b/packages/coding-agent/src/session/stream-guards.ts new file mode 100644 index 000000000..6f9996d7f --- /dev/null +++ b/packages/coding-agent/src/session/stream-guards.ts @@ -0,0 +1,417 @@ +import * as fs from "node:fs"; +import type { Agent, AgentEvent, AgentMessage, AgentTurnEndContext } from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage, AssistantMessageEvent, Model, ToolCall } from "@oh-my-pi/pi-ai"; +import { GeminiHeaderRunDetector, isGeminiThinkingModel } from "@oh-my-pi/pi-ai/utils/thinking-loop"; +import { type RepeatedToolCallDetection, ToolCallLoopGuard } from "@oh-my-pi/pi-ai/utils/tool-call-loop-guard"; +import { isEnoent, logger, prompt } from "@oh-my-pi/pi-utils"; +import type { Settings } from "../config/settings"; +import { normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit"; +import { type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls"; +import geminiToolReminderTemplate from "../prompts/system/gemini-tool-call-reminder.md" with { type: "text" }; +import toolCallLoopRedirectTemplate from "../prompts/system/tool-call-loop-redirect.md" with { type: "text" }; +import type { SecretObfuscator } from "../secrets/obfuscator"; +import { assertEditableFile } from "../tools/auto-generated-guard"; +import { isInternalUrlPath, normalizeLocalScheme, resolveToCwd } from "../tools/path-utils"; +import { ToolError } from "../tools/tool-errors"; +import type { CustomMessage } from "./messages"; +import type { SessionManager } from "./session-manager"; + +const GEMINI_HEADER_INTERRUPT_REASON = "Interrupted: emit a tool call instead of more planning"; +const GEMINI_TOOL_REMINDER_TYPE = "gemini-tool-call-reminder"; +const TOOL_CALL_LOOP_REDIRECT_TYPE = "tool-call-loop-redirect"; + +/** Capabilities borrowed by the session's streaming and loop guards. */ +export interface StreamGuardsHost { + agent: Agent; + settings: Settings; + sessionManager: SessionManager; + obfuscator: SecretObfuscator | undefined; + model(): Model | undefined; + isDisposed(): boolean; + promptGeneration(): number; + localProtocolOptions(): LocalProtocolOptions; + emitNotice(level: "info" | "warning" | "error", message: string, source?: string): void; + schedulePostPromptTask(task: (signal: AbortSignal) => Promise): void; + discardAssistantTurn(message: AssistantMessage): void; +} + +/** Guards streamed edit calls against generated files and invalid patch previews. */ +export class StreamingEditGuard { + readonly #host: StreamGuardsHost; + #abortTriggered = false; + #checkedLineCounts = new Map(); + #precheckedToolCallIds = new Set(); + #fileCache = new Map(); + #lastToolCallId: string | undefined; + + constructor(host: StreamGuardsHost) { + this.#host = host; + } + + /** Whether the current turn was aborted by streaming edit validation. */ + get abortTriggered(): boolean { + return this.#abortTriggered; + } + + /** Clears all turn-scoped streaming edit state. */ + reset(): void { + this.#abortTriggered = false; + this.#checkedLineCounts.clear(); + this.#precheckedToolCallIds.clear(); + this.#fileCache.clear(); + } + + /** Pre-caches and validates a streamed edit as its arguments arrive. */ + preCache(event: AgentEvent): void { + if (this.#abortTriggered || event.type !== "message_update") return; + const assistantEvent = event.assistantMessageEvent; + if ( + assistantEvent.type !== "toolcall_start" && + assistantEvent.type !== "toolcall_delta" && + assistantEvent.type !== "toolcall_end" + ) { + return; + } + const streamingEdit = this.#getToolCall(event); + if (!streamingEdit) return; + + // The auto-generated guard runs unconditionally: editing a generated file + // is never the user's intent, and the cost of a false-positive abort is one + // wasted turn vs. silently corrupting a regenerated source. + const shouldCheckAutoGenerated = + !streamingEdit.toolCall.id || !this.#precheckedToolCallIds.has(streamingEdit.toolCall.id); + if (shouldCheckAutoGenerated) { + if (streamingEdit.toolCall.id) this.#precheckedToolCallIds.add(streamingEdit.toolCall.id); + this.#abortForAutoGeneratedPath(streamingEdit.toolCall, streamingEdit.path, streamingEdit.resolvedPath); + } + + // File-cache priming feeds maybeAbort's removed-lines check, which is the + // optional patch-preview verification gated by edit.streamingAbort. + if (this.#host.settings.get("edit.streamingAbort")) this.#ensureFileCache(streamingEdit.resolvedPath); + } + + /** Invalidates cached source text after an edit tool result lands. */ + invalidate(filePath: string): void { + const resolvedPath = this.#resolveSessionFsPath(filePath); + if (resolvedPath !== undefined) this.#fileCache.delete(resolvedPath); + } + + /** Aborts a streamed edit whose completed patch preview cannot apply. */ + maybeAbort(event: AgentEvent): void { + if (!this.#host.settings.get("edit.streamingAbort") || this.#abortTriggered || event.type !== "message_update") { + return; + } + const assistantEvent = event.assistantMessageEvent; + if (assistantEvent.type !== "toolcall_end" && assistantEvent.type !== "toolcall_delta") return; + const streamingEdit = this.#getToolCall(event); + if (!streamingEdit?.toolCall.id) return; + + const { toolCall, path, resolvedPath, diff, op, rename } = streamingEdit; + if (!diff || (op && op !== "update") || !diff.includes("\n")) return; + const lastNewlineIndex = diff.lastIndexOf("\n"); + if (lastNewlineIndex < 0) return; + const diffForCheck = diff.endsWith("\n") ? diff : diff.slice(0, lastNewlineIndex + 1); + if (diffForCheck.trim().length === 0) return; + + let normalizedDiff = normalizeDiff(diffForCheck.replace(/\r/g, "")); + if (!normalizedDiff) return; + if (this.#host.obfuscator) normalizedDiff = this.#host.obfuscator.deobfuscate(normalizedDiff); + if (!normalizedDiff) return; + const lines = normalizedDiff.split("\n"); + if (!lines.some(line => line.startsWith("+") || line.startsWith("-"))) return; + + const lineCount = lines.length; + const lastChecked = this.#checkedLineCounts.get(toolCall.id); + if (lastChecked !== undefined && lineCount <= lastChecked) return; + this.#checkedLineCounts.set(toolCall.id, lineCount); + + const removedLines = lines + .filter(line => line.startsWith("-") && !line.startsWith("--- ")) + .map(line => line.slice(1)); + if (removedLines.length > 0) { + let cachedContent = this.#fileCache.get(resolvedPath); + if (cachedContent === undefined) { + this.#ensureFileCache(resolvedPath); + cachedContent = this.#fileCache.get(resolvedPath); + } + if (cachedContent !== undefined) { + const missing = removedLines.find(line => !cachedContent.includes(normalizeToLF(line))); + if (missing) this.#abortPatch(toolCall.id, path, `Failed to find expected lines in ${path}:\n${missing}`); + return; + } + if (assistantEvent.type === "toolcall_delta") return; + void this.#checkRemovedLines(toolCall.id, path, resolvedPath, removedLines); + return; + } + if (assistantEvent.type === "toolcall_delta") return; + void this.#checkPreviewPatch(toolCall.id, path, rename, normalizedDiff); + } + + #getToolCall(event: AgentEvent): + | { + toolCall: ToolCall; + path: string; + resolvedPath: string; + diff?: string; + op?: string; + rename?: string; + } + | undefined { + if (event.type !== "message_update" || event.message.role !== "assistant") return undefined; + const contentIndex = event.assistantMessageEvent.contentIndex ?? 0; + const messageContent = event.message.content; + if (!Array.isArray(messageContent) || contentIndex < 0 || contentIndex >= messageContent.length) return undefined; + const toolCall = messageContent[contentIndex] as ToolCall; + if (toolCall.name !== "edit") return undefined; + const args = toolCall.arguments; + if (!args || typeof args !== "object" || Array.isArray(args) || "old_text" in args || "new_text" in args) { + return undefined; + } + const filePath = typeof args.path === "string" ? args.path : undefined; + if (!filePath) return undefined; + // local:// URLs resolve to artifacts; other internal URLs have no local path. + const resolvedPath = this.#resolveSessionFsPath(filePath); + if (resolvedPath === undefined) return undefined; + return { + toolCall, + path: filePath, + resolvedPath, + diff: typeof args.diff === "string" ? args.diff : undefined, + op: typeof args.op === "string" ? args.op : undefined, + rename: typeof args.rename === "string" ? args.rename : undefined, + }; + } + + #abortForAutoGeneratedPath(toolCall: ToolCall, filePath: string, resolvedPath: string): void { + if (this.#lastToolCallId === toolCall.id) return; + this.#lastToolCallId = toolCall.id; + void assertEditableFile(resolvedPath, filePath).catch(error => { + if (!(error instanceof ToolError) || this.#lastToolCallId !== toolCall.id) return; + if (!this.#abortTriggered) { + this.#abortTriggered = true; + logger.warn("Streaming edit aborted due to auto-generated file guard", { + toolCallId: toolCall.id, + path: filePath, + }); + this.#host.agent.abort(); + } + }); + } + + #ensureFileCache(resolvedPath: string): void { + if (this.#fileCache.has(resolvedPath)) return; + try { + const rawText = fs.readFileSync(resolvedPath, "utf-8"); + const { text } = stripBom(rawText); + this.#fileCache.set(resolvedPath, normalizeToLF(text)); + } catch { + // Read errors are handled by the edit tool itself. + } + } + + #resolveSessionFsPath(filePath: string): string | undefined { + const normalized = normalizeLocalScheme(filePath); + if (normalized.startsWith("local:")) { + return resolveLocalUrlToPath(normalized, this.#host.localProtocolOptions()); + } + if (isInternalUrlPath(normalized)) return undefined; + return resolveToCwd(normalized, this.#host.sessionManager.getCwd()); + } + + async #checkRemovedLines( + toolCallId: string, + filePath: string, + resolvedPath: string, + removedLines: string[], + ): Promise { + if (this.#abortTriggered) return; + try { + const { text } = stripBom(await Bun.file(resolvedPath).text()); + const normalizedContent = normalizeToLF(text); + const missing = removedLines.find(line => !normalizedContent.includes(normalizeToLF(line))); + if (missing) + this.#abortPatch(toolCallId, filePath, `Failed to find expected lines in ${filePath}:\n${missing}`); + } catch (error) { + if (!isEnoent(error)) { + // Unexpected fallback read errors remain non-fatal. + } + } + } + + async #checkPreviewPatch( + toolCallId: string, + filePath: string, + rename: string | undefined, + normalizedDiff: string, + ): Promise { + if (this.#abortTriggered) return; + try { + await previewPatch( + { path: filePath, op: "update", rename, diff: normalizedDiff }, + { + cwd: this.#host.sessionManager.getCwd(), + allowFuzzy: this.#host.settings.get("edit.fuzzyMatch"), + fuzzyThreshold: this.#host.settings.get("edit.fuzzyThreshold"), + }, + ); + } catch (error) { + if (error instanceof ParseError) return; + this.#abortPatch(toolCallId, filePath, error instanceof Error ? error.message : String(error)); + } + } + + #abortPatch(toolCallId: string, filePath: string, error: string): void { + this.#abortTriggered = true; + logger.warn("Streaming edit aborted due to patch preview failure", { toolCallId, path: filePath, error }); + this.#host.agent.abort(); + } +} + +/** Detects cross-turn tool loops and Gemini reasoning-header runaways. */ +export class LoopGuards { + readonly #host: StreamGuardsHost; + #geminiHeaderDetector: GeminiHeaderRunDetector | undefined; + #toolCallLoopGuard: ToolCallLoopGuard | undefined; + #toolCallLoopGuardSettingsKey: string | undefined; + + constructor(host: StreamGuardsHost) { + this.#host = host; + } + + /** Records a completed turn and injects a redirect when calls repeat. */ + recordTurn(messages: AgentMessage[], context: AgentTurnEndContext | undefined): void { + if (context?.message.role !== "assistant") return; + const detection = this.#activeToolCallLoopGuard()?.recordTurn({ + message: context.message, + toolResults: context.toolResults, + }); + if (detection) this.#injectToolCallLoopRedirect(messages, detection); + } + + /** Feeds a streamed assistant event to the Gemini header-runaway detector. */ + onAssistantEvent(message: AssistantMessage, event: AssistantMessageEvent): void { + if (event.type === "thinking_start") { + this.#geminiHeaderDetector = this.#geminiHeaderGuardActive() ? new GeminiHeaderRunDetector() : undefined; + return; + } + const detector = this.#geminiHeaderDetector; + if (!detector) return; + if (event.type === "thinking_delta") { + if (detector.push(event.delta)) this.#interruptGeminiHeaderRunaway(detector.count, message.timestamp); + return; + } + if (event.type === "text_start" || event.type === "toolcall_start") detector.reset(); + } + + #activeToolCallLoopGuard(): ToolCallLoopGuard | undefined { + if (this.#host.settings.get("model.toolCallLoopGuard.enabled") !== true) { + this.#toolCallLoopGuard = undefined; + this.#toolCallLoopGuardSettingsKey = undefined; + return undefined; + } + const threshold = this.#host.settings.get("model.toolCallLoopGuard.threshold"); + const exemptTools = this.#host.settings + .get("model.toolCallLoopGuard.exemptTools") + .filter((tool): tool is string => typeof tool === "string" && tool.length > 0); + const settingsKey = `${threshold}:${JSON.stringify(exemptTools)}`; + if (!this.#toolCallLoopGuard || this.#toolCallLoopGuardSettingsKey !== settingsKey) { + this.#toolCallLoopGuard = new ToolCallLoopGuard({ threshold, exemptTools }); + this.#toolCallLoopGuardSettingsKey = settingsKey; + } + return this.#toolCallLoopGuard; + } + + #injectToolCallLoopRedirect(messages: AgentMessage[], detection: RepeatedToolCallDetection): void { + const content = prompt.render(toolCallLoopRedirectTemplate, { + tool_name: detection.toolName, + count: detection.count, + arguments_summary: detection.argumentsSummary, + result_summary: detection.resultSummary || "(no text result)", + }); + const details = { + toolName: detection.toolName, + count: detection.count, + argumentsSummary: detection.argumentsSummary, + resultSummary: detection.resultSummary, + }; + logger.warn("cross-turn tool-call loop detected", { toolName: detection.toolName, count: detection.count }); + const redirectMessage: CustomMessage = { + role: "custom", + customType: TOOL_CALL_LOOP_REDIRECT_TYPE, + content, + display: false, + details, + attribution: "agent", + timestamp: Date.now(), + }; + messages.push(redirectMessage); + if (this.#host.agent.state.messages !== messages) this.#host.agent.appendMessage(redirectMessage); + this.#host.sessionManager.appendCustomMessageEntry( + TOOL_CALL_LOOP_REDIRECT_TYPE, + content, + false, + details, + "agent", + ); + } + + #geminiHeaderGuardActive(): boolean { + const model = this.#host.model(); + return ( + process.env.PI_NO_THINKING_LOOP_GUARD !== "1" && + this.#host.settings.get("model.loopGuard.enabled") === true && + this.#host.settings.get("model.loopGuard.toolCallReminder") === true && + model !== undefined && + isGeminiThinkingModel(model) + ); + } + + #interruptGeminiHeaderRunaway(headerCount: number, targetTimestamp: number): void { + const model = this.#host.model(); + logger.warn("Gemini reasoning-header runaway; interrupting to require a tool call", { + model: model?.id, + provider: model?.provider, + headers: headerCount, + }); + this.#host.emitNotice( + "warning", + `Interrupted ${headerCount} planning headers with no tool call; reminded the model to issue one.`, + "loop-guard", + ); + this.#host.agent.abort(GEMINI_HEADER_INTERRUPT_REASON); + const generation = this.#host.promptGeneration(); + this.#host.schedulePostPromptTask(async signal => { + if (signal.aborted || this.#host.isDisposed() || this.#host.promptGeneration() !== generation) return; + await this.#host.agent.waitForIdle(); + if (signal.aborted || this.#host.isDisposed() || this.#host.promptGeneration() !== generation) return; + const aborted = this.#host.agent.state.messages.findLast( + (message): message is AssistantMessage => + message.role === "assistant" && message.timestamp === targetTimestamp, + ); + if (aborted) this.#host.discardAssistantTurn(aborted); + const content = prompt.render(geminiToolReminderTemplate, { count: headerCount }); + const details = { headers: headerCount }; + this.#host.agent.appendMessage({ + role: "custom", + customType: GEMINI_TOOL_REMINDER_TYPE, + content, + display: false, + details, + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.sessionManager.appendCustomMessageEntry( + GEMINI_TOOL_REMINDER_TYPE, + content, + false, + details, + "agent", + ); + try { + await this.#host.agent.continue(); + } catch (error) { + logger.warn("gemini tool-call reminder continue failed", { error: String(error) }); + } + }); + } +} diff --git a/packages/coding-agent/src/session/todo-tracker.ts b/packages/coding-agent/src/session/todo-tracker.ts new file mode 100644 index 000000000..8174cc100 --- /dev/null +++ b/packages/coding-agent/src/session/todo-tracker.ts @@ -0,0 +1,380 @@ +import type { Agent, AgentMessage, AgentTool } from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage, Message, Model, TextContent, ToolChoice } from "@oh-my-pi/pi-ai"; +import { isRecord, logger, prompt, stringProperty } from "@oh-my-pi/pi-utils"; +import type { Settings } from "../config/settings"; +import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" }; +import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" }; +import midRunTodoNudgePrompt from "../prompts/system/mid-run-todo-nudge.md" with { type: "text" }; +import { getLatestTodoPhasesFromEntries, isTodoPhase, type TodoItem, type TodoPhase } from "../tools/todo"; +import { buildNamedToolChoice } from "../utils/tool-choice"; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { SessionManager } from "./session-manager"; + +const MID_RUN_NUDGE_MUTATION_THRESHOLD = 12; +const MID_RUN_NUDGE_MAX_PER_CYCLE = 2; +const MUTATING_TOOLS: Record = { + bash: true, + eval: true, + edit: true, + write: true, + ast_edit: true, +}; +const MID_RUN_NUDGE_MESSAGE_TYPE = "mid-run-todo-nudge"; +const MARKDOWN_PROMPT_PREFIX_RE = /^(?:>\s*)?(?:(?:[-*+]|\d+[.)])\s+)*/; +const PROMPT_LABEL_RE = /^(?:q(?:uestion)?|ask)\s*\d*\s*[:.)-]\s*/i; +const QUESTION_PROMPT_RE = + /^(?:what|which|when|where|why|how|who|whom|whose|do|does|did|can|could|would|will|should|is|are|am|may|shall)\b/i; +const USER_DIRECTED_PROMPT_RE = /\b(?:you|your|we|our)\b/i; +const USER_RESPONSE_CUE_RE = + /^(?:please\s+)?(?:confirm|reply|choose|pick|decide|advise)\b|^(?:please\s+)?answer\b|^(?:please\s+)?(?:let\s+me\s+know|tell\s+me)\b/i; + +interface PromptLine { + text: string; + hadPromptLabel: boolean; +} + +/** Capabilities the todo tracker borrows from its owning session. */ +export interface TodoTrackerHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + model(): Model | undefined; + agentKind(): "main" | "sub"; + emitSessionEvent(event: AgentSessionEvent): Promise; + scheduleAgentContinue(options: { generation?: number }): void; + promptGeneration(): number; + hasPendingAsyncWake(): boolean; + getActiveToolNames(): string[]; + toolRegistry(): Map; + planModeEnabled(): boolean; + consumeLastServedToolChoiceLabel(): string | undefined; +} + +/** Owns canonical todo state, eager preludes, and completion reminders. */ +export class TodoTracker { + readonly #host: TodoTrackerHost; + #phases: TodoPhase[] = []; + #reminderCount = 0; + #reminderAwaitingProgress = false; + #mutationsSinceLastTouch = 0; + #midRunNudgeCount = 0; + + constructor(host: TodoTrackerHost) { + this.#host = host; + } + + /** Returns a defensive clone of the current todo phases. */ + get phases(): TodoPhase[] { + return this.#clonePhases(this.#phases); + } + + /** Replaces todo phases with a defensive clone. */ + setPhases(phases: TodoPhase[]): void { + this.#phases = this.#clonePhases(phases); + } + + /** Rehydrates todo phases from the current transcript branch. */ + syncFromBranch(): void { + this.setPhases(getLatestTodoPhasesFromEntries(this.#host.sessionManager.getBranch())); + } + + /** Returns a defensive clone suitable for snapshots and branch state. */ + clonePhases(phases: TodoPhase[]): TodoPhase[] { + return this.#clonePhases(phases); + } + + /** Resets per-prompt reminder and mutation budgets. */ + resetCycle(): void { + this.#reminderCount = 0; + this.#reminderAwaitingProgress = false; + this.#mutationsSinceLastTouch = 0; + this.#midRunNudgeCount = 0; + } + + /** Records a completed tool result before asynchronous event processing begins. */ + onToolResult(toolName: string, isError: boolean): void { + if (toolName === "todo") { + this.#mutationsSinceLastTouch = 0; + } else if (!isError && MUTATING_TOOLS[toolName]) { + this.#mutationsSinceLastTouch++; + } + this.#reminderAwaitingProgress = false; + } + + /** Detects whether a successful todo result came from an init operation. */ + onTodoResultDetails(details: Record, toolCallId: string | undefined): boolean { + const phases = details.phases; + if (!Array.isArray(phases) || !phases.every(isTodoPhase)) return false; + const detailOp = stringProperty(details, "op"); + if (detailOp) return detailOp === "init"; + if (!toolCallId) return false; + for (let index = this.#host.agent.state.messages.length - 1; index >= 0; index--) { + const message = this.#host.agent.state.messages[index]; + if (!message) continue; + const op = toolCallOpFromMessage(message, toolCallId); + if (op) return op === "init"; + } + return false; + } + + /** Builds the first-turn eager todo prelude and optional forced tool choice. */ + createEagerTodoPrelude( + promptText: string | undefined, + ): { message: AgentMessage; toolChoice?: ToolChoice } | undefined { + const mode = this.#host.settings.get("todo.eager"); + if (mode === "default" || !this.#host.settings.get("todo.enabled")) return undefined; + if (this.#host.planModeEnabled() || this.#phases.length > 0) return undefined; + if (promptText !== undefined) { + if (this.#host.agent.state.messages.some(message => message.role === "user")) return undefined; + const trimmedPromptText = promptText.trimEnd(); + if (trimmedPromptText.endsWith("?") || trimmedPromptText.endsWith("!")) return undefined; + } + const activeToolNames = this.#host.getActiveToolNames(); + if (!activeToolNames.includes("todo")) { + logger.warn("Eager todo enforcement skipped because todo is not active", { activeToolNames }); + return undefined; + } + const message: AgentMessage = { + role: "custom", + customType: "eager-todo-prelude", + content: prompt.render(eagerTodoPrompt, { ...this.#buildEagerPreludeContext(), forced: mode === "always" }), + display: false, + attribution: "agent", + timestamp: Date.now(), + }; + if (promptText === undefined || mode === "preferred") return { message }; + const model = this.#host.model(); + const toolChoice = buildNamedToolChoice("todo", model); + if (!toolChoice) { + logger.warn( + "Eager todo proceeding with the reminder only because the current model does not support a forced todo tool_choice", + { modelApi: model?.api, modelId: model?.id }, + ); + return { message }; + } + return { message, toolChoice }; + } + + /** Builds the first-turn eager task-delegation prelude. */ + createEagerTaskPrelude(promptText: string | undefined): AgentMessage | undefined { + if (this.#host.settings.get("task.eager") !== "always") return undefined; + if (this.#host.agentKind() === "sub" || this.#host.planModeEnabled()) return undefined; + if (promptText !== undefined) { + if (this.#host.agent.state.messages.some(message => message.role === "user")) return undefined; + const trimmed = promptText.trimEnd(); + if (trimmed.endsWith("?") || trimmed.endsWith("!")) return undefined; + } + if (!this.#host.getActiveToolNames().includes("task")) return undefined; + return { + role: "custom", + customType: "eager-task-prelude", + content: prompt.render(eagerTaskPrompt, this.#buildEagerPreludeContext()), + display: false, + attribution: "agent", + timestamp: Date.now(), + }; + } + + /** Builds reminder-only eager preludes after compaction. */ + buildPostCompactionEagerNudges(): AgentMessage[] { + const nudges: AgentMessage[] = []; + const todo = this.createEagerTodoPrelude(undefined); + if (todo) nudges.push(todo.message); + const task = this.createEagerTaskPrelude(undefined); + if (task) nudges.push(task); + return nudges; + } + + /** Checks a terminal assistant turn and schedules continuation for incomplete todos. */ + async checkCompletion(message: AssistantMessage): Promise { + if (this.#host.consumeLastServedToolChoiceLabel() === "user-force") return false; + if (this.#host.planModeEnabled()) return false; + if (this.#reminderAwaitingProgress) { + logger.debug("Todo completion: prior reminder still awaiting agent action; staying silent", { + attempt: this.#reminderCount, + }); + return false; + } + if (!this.#host.settings.get("todo.reminders") || !this.#host.settings.get("todo.enabled")) { + this.#reminderCount = 0; + this.#reminderAwaitingProgress = false; + return false; + } + const remindersMax = this.#host.settings.get("todo.remindersMax"); + if (this.#reminderCount >= remindersMax) { + logger.debug("Todo completion: max reminders reached", { count: this.#reminderCount }); + return false; + } + const phases = this.phases; + if (phases.length === 0) { + this.#reminderCount = 0; + this.#reminderAwaitingProgress = false; + return false; + } + const incompleteByPhase = phases + .map(phase => ({ + name: phase.name, + tasks: phase.tasks + .filter( + (task): task is TodoItem & { status: "pending" | "in_progress" } => + task.status === "pending" || task.status === "in_progress", + ) + .map(task => ({ content: task.content, status: task.status })), + })) + .filter(phase => phase.tasks.length > 0); + const incomplete = incompleteByPhase.flatMap(phase => phase.tasks); + if (incomplete.length === 0) { + this.#reminderCount = 0; + this.#reminderAwaitingProgress = false; + return false; + } + if (isAwaitingUserAnswer(message)) { + logger.debug("Todo completion: assistant is waiting for user input; skipping reminder", { + incomplete: incomplete.length, + }); + return false; + } + if (this.#host.hasPendingAsyncWake()) { + logger.debug("Todo completion: async jobs in flight will re-wake the loop; skipping reminder", { + incomplete: incomplete.length, + }); + return false; + } + this.#reminderCount++; + const todoList = incompleteByPhase + .map(phase => `- ${phase.name}\n${phase.tasks.map(task => ` - ${task.content}`).join("\n")}`) + .join("\n"); + const reminder = + `\n` + + `You stopped with ${incomplete.length} incomplete todo item(s):\n${todoList}\n\n` + + `Please continue working on these tasks or mark them complete if finished.\n` + + `(Reminder ${this.#reminderCount}/${remindersMax})\n` + + ``; + logger.debug("Todo completion: sending reminder", { + incomplete: incomplete.length, + attempt: this.#reminderCount, + }); + await this.#host.emitSessionEvent({ + type: "todo_reminder", + todos: incomplete, + attempt: this.#reminderCount, + maxAttempts: remindersMax, + }); + const reminderMessage: Message = { + role: "developer", + content: [{ type: "text", text: reminder }], + attribution: "agent", + timestamp: Date.now(), + }; + this.#mutationsSinceLastTouch = 0; + this.#reminderAwaitingProgress = true; + this.#host.agent.appendMessage(reminderMessage); + this.#host.sessionManager.appendMessage(reminderMessage); + this.#host.scheduleAgentContinue({ generation: this.#host.promptGeneration() }); + return true; + } + + /** Takes the next hidden mid-run reconciliation nudge, if its budget and guards allow. */ + takeMidRunNudge(): AgentMessage | null { + if (this.#mutationsSinceLastTouch < MID_RUN_NUDGE_MUTATION_THRESHOLD) return null; + if (this.#midRunNudgeCount >= MID_RUN_NUDGE_MAX_PER_CYCLE) return null; + if (!this.#host.settings.get("todo.enabled") || !this.#host.settings.get("todo.reminders")) return null; + if (this.#host.planModeEnabled() || !this.#host.getActiveToolNames().includes("todo")) return null; + const incomplete = this.#phases + .flatMap(phase => phase.tasks) + .filter(task => task.status === "pending" || task.status === "in_progress"); + if (incomplete.length === 0) return null; + this.#mutationsSinceLastTouch = 0; + this.#midRunNudgeCount++; + const { toolRefs } = this.#buildEagerPreludeContext(); + const reminder = prompt.render(midRunTodoNudgePrompt, { + toolRefs, + incompleteCount: incomplete.length, + plural: incomplete.length !== 1, + }); + logger.debug("Mid-run todo nudge fired", { + incomplete: incomplete.length, + nudge: this.#midRunNudgeCount, + }); + return { + role: "custom", + customType: MID_RUN_NUDGE_MESSAGE_TYPE, + content: reminder, + display: false, + attribution: "agent", + timestamp: Date.now(), + }; + } + + #buildEagerPreludeContext(): { toolRefs: Record; taskBatch: boolean } { + const wireName = (name: string): string => { + const tool = this.#host.toolRegistry().get(name); + return typeof tool?.customWireName === "string" ? tool.customWireName : name; + }; + return { + toolRefs: { task: wireName("task"), todo: wireName("todo") }, + taskBatch: this.#host.settings.get("task.batch"), + }; + } + + #clonePhases(phases: TodoPhase[]): TodoPhase[] { + return phases.map(phase => ({ + name: phase.name, + tasks: phase.tasks.map(task => + task.blocker !== undefined + ? { content: task.content, status: task.status, blocker: task.blocker } + : { content: task.content, status: task.status }, + ), + })); + } +} + +function toolCallOpFromMessage(message: AgentMessage, toolCallId: string): string | undefined { + if (message.role !== "assistant" || !Array.isArray(message.content)) return undefined; + for (const block of message.content) { + if (!isRecord(block) || block.type !== "toolCall" || block.id !== toolCallId) continue; + return isRecord(block.arguments) ? stringProperty(block.arguments, "op") : undefined; + } + return undefined; +} + +function assistantText(message: AssistantMessage): string { + return message.content + .filter((content): content is TextContent => content.type === "text") + .map(content => content.text) + .join("\n") + .trim(); +} + +function promptLine(line: string): PromptLine { + const withoutMarkdownPrefix = line.trim().replace(MARKDOWN_PROMPT_PREFIX_RE, "").trim(); + const withoutPromptLabel = withoutMarkdownPrefix.replace(PROMPT_LABEL_RE, "").trim(); + return { + text: withoutPromptLabel, + hadPromptLabel: withoutPromptLabel !== withoutMarkdownPrefix, + }; +} + +function isQuestionPromptLine(line: string): boolean { + const candidate = promptLine(line); + if (!/[??]\s*$/.test(candidate.text)) return false; + return ( + candidate.hadPromptLabel || + QUESTION_PROMPT_RE.test(candidate.text) || + USER_DIRECTED_PROMPT_RE.test(candidate.text) + ); +} + +function isResponseCueLine(line: string): boolean { + const candidate = promptLine(line) + .text.replace(/[.!?。!?]+$/, "") + .trim(); + return USER_RESPONSE_CUE_RE.test(candidate); +} + +function isAwaitingUserAnswer(message: AssistantMessage): boolean { + const text = assistantText(message); + if (!text) return false; + const lastLine = text.split(/\r?\n/).at(-1)?.trim(); + return lastLine !== undefined && (isQuestionPromptLine(lastLine) || isResponseCueLine(lastLine)); +} diff --git a/packages/coding-agent/src/session/ttsr-coordinator.ts b/packages/coding-agent/src/session/ttsr-coordinator.ts new file mode 100644 index 000000000..a2929a3f5 --- /dev/null +++ b/packages/coding-agent/src/session/ttsr-coordinator.ts @@ -0,0 +1,496 @@ +import * as os from "node:os"; +import * as path from "node:path"; +import { + type AfterToolCallContext, + type AfterToolCallResult, + type Agent, + type AgentEvent, + type AgentMessage, + createToolScopedAbortReason, +} from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage, ToolCall } from "@oh-my-pi/pi-ai"; +import { isRecord, prompt, relativePathWithinRoot } from "@oh-my-pi/pi-utils"; +import type { Rule } from "../capability/rule"; +import type { Settings } from "../config/settings"; +import type { TtsrManager, TtsrMatchContext } from "../export/ttsr"; +import ttsrInterruptTemplate from "../prompts/system/ttsr-interrupt.md" with { type: "text" }; +import ttsrToolReminderTemplate from "../prompts/system/ttsr-tool-reminder.md" with { type: "text" }; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { SessionManager } from "./session-manager"; + +interface TtsrContinueOptions { + delayMs?: number; + generation?: number; + shouldContinue?: () => boolean; + onSkip?: () => void; + onError?: () => void; +} + +/** Capabilities the TTSR coordinator borrows from its owning session. */ +export interface TtsrCoordinatorHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + emitSessionEvent(event: AgentSessionEvent): Promise; + schedulePostPromptTask(task: (signal: AbortSignal) => Promise, options?: { delayMs?: number }): void; + scheduleAgentContinue(options: TtsrContinueOptions): void; + promptGeneration(): number; +} + +/** Coordinates TTSR stream matching, interruption, injection, and resume gates. */ +export class TtsrCoordinator { + readonly #host: TtsrCoordinatorHost; + readonly #manager: TtsrManager | undefined; + #pendingInjections: Rule[] = []; + #perToolInjections = new Map(); + #abortPending = false; + #retryToken = 0; + #resumePromise: Promise | undefined; + #resumeResolve: (() => void) | undefined; + + constructor(host: TtsrCoordinatorHost, manager: TtsrManager | undefined) { + this.#host = host; + this.#manager = manager; + } + + /** Configured TTSR manager, when stream rules are enabled. */ + get manager(): TtsrManager | undefined { + return this.#manager; + } + + /** Whether a TTSR-triggered stream abort is awaiting its continuation. */ + get abortPending(): boolean { + return this.#abortPending; + } + + /** Current resume gate awaited by post-prompt recovery. */ + get resumeGate(): Promise | undefined { + return this.#resumePromise; + } + + /** Resets stream buffers at turn start. */ + onTurnStart(): void { + this.#manager?.resetBuffer(); + } + + /** Advances repeat-after-gap tracking at turn end. */ + onTurnEnd(): void { + this.#manager?.incrementMessageCount(); + } + + /** Checks one streamed message update and reports whether TTSR consumed it by aborting. */ + async checkMessageUpdate(event: AgentEvent): Promise { + if (event.type !== "message_update" || !this.#manager?.hasRules()) return false; + const assistantEvent = event.assistantMessageEvent; + let matchContext: TtsrMatchContext | undefined; + let streamingToolCall: ToolCall | undefined; + if (assistantEvent.type === "text_delta") { + matchContext = { source: "text" }; + } else if (assistantEvent.type === "thinking_delta") { + matchContext = { source: "thinking" }; + } else if (assistantEvent.type === "toolcall_delta") { + streamingToolCall = this.#getStreamingToolCallBlock(event.message, assistantEvent.contentIndex); + matchContext = this.#getToolMatchContext(streamingToolCall, assistantEvent.contentIndex); + } + if (!matchContext || !("delta" in assistantEvent)) return false; + const targetMessageTimestamp = event.message.role === "assistant" ? event.message.timestamp : undefined; + const matches = this.#checkStream(assistantEvent.delta, matchContext, streamingToolCall); + if (matches.length > 0 && this.#handleMatches(matches, matchContext, targetMessageTimestamp)) return true; + // AST rules use the reconstructed edit/write snapshot and are awaited so + // the manager self-throttles native matching. + if (matchContext.source === "tool" && this.#manager.hasAstRules()) { + const astMatches = await this.#checkAstStream(matchContext, streamingToolCall); + if (astMatches.length > 0 && this.#handleMatches(astMatches, matchContext, targetMessageTimestamp)) + return true; + } + return false; + } + + /** Settles the previous resume gate and queues any deferred injection. */ + onAssistantMessageEnd(message: AssistantMessage): void { + // Gate on abortPending, not stopReason: unrelated aborts have no TTSR continuation. + if (!this.#abortPending) this.resolveResume(); + this.#queueDeferredInjectionIfNeeded(message); + } + + /** Marks names persisted with a delivered TTSR injection as injected. */ + markInjectedFromDetails(details: unknown): void { + if (!details || typeof details !== "object" || Array.isArray(details)) return; + const rules = "rules" in details ? details.rules : undefined; + if (!Array.isArray(rules)) return; + this.#markInjected(rules.filter((ruleName): ruleName is string => typeof ruleName === "string")); + } + + /** Folds per-tool reminders into the matched tool's result. */ + afterToolCall(ctx: AfterToolCallContext): AfterToolCallResult | undefined { + const rules = this.#perToolInjections.get(ctx.toolCall.id); + if (!rules || rules.length === 0) return undefined; + this.#perToolInjections.delete(ctx.toolCall.id); + const reminder = rules + .map(rule => + prompt.render(ttsrToolReminderTemplate, { + name: rule.name, + path: this.#displayRulePath(rule.path), + content: rule.content, + }), + ) + .join("\n\n"); + const ruleNames = rules.map(rule => rule.name.trim()).filter(name => name.length > 0); + if (ruleNames.length > 0) this.#host.sessionManager.appendTtsrInjection(ruleNames); + return { content: [{ type: "text", text: reminder }, ...ctx.result.content] }; + } + + /** Resolves and clears the current resume gate. */ + resolveResume(): void { + if (!this.#resumeResolve) return; + this.#resumeResolve(); + this.#resumeResolve = undefined; + this.#resumePromise = undefined; + } + + #ensureResumePromise(): void { + if (this.#resumePromise) return; + const { promise, resolve } = Promise.withResolvers(); + this.#resumePromise = promise; + this.#resumeResolve = resolve; + } + + #formatAbortReason(rules: Rule[]): string { + const label = rules.length === 1 ? "rule" : "rules"; + return `TTSR matched ${label}: ${rules.map(rule => rule.name).join(", ")}`; + } + + #getInjectionContent(): { content: string; rules: Rule[] } | undefined { + if (this.#pendingInjections.length === 0) return undefined; + const rules = this.#pendingInjections; + const content = rules + .map(rule => + prompt.render(ttsrInterruptTemplate, { + name: rule.name, + path: this.#displayRulePath(rule.path), + content: rule.content, + }), + ) + .join("\n\n"); + this.#pendingInjections = []; + return { content, rules }; + } + + #displayRulePath(rulePath: string): string { + const cwd = this.#host.sessionManager.getCwd(); + const cwdRelative = relativePathWithinRoot(cwd, rulePath) ?? this.#displayPathWithinRoot(cwd, rulePath); + if (cwdRelative) return cwdRelative; + const homeRelative = relativePathWithinRoot(os.homedir(), rulePath); + if (homeRelative) return `~/${homeRelative}`; + return rulePath; + } + + #displayPathWithinRoot(root: string, candidate: string): string | null { + const relative = path.relative(path.resolve(root), path.resolve(candidate)); + return relative && !relative.startsWith("..") && !path.isAbsolute(relative) ? relative : null; + } + + #addPendingInjections(rules: Rule[]): void { + const seen = new Set(this.#pendingInjections.map(rule => rule.name)); + for (const rule of rules) { + if (seen.has(rule.name)) continue; + this.#pendingInjections.push(rule); + seen.add(rule.name); + } + } + + #extractToolCallId(matchContext: TtsrMatchContext): string | undefined { + if (matchContext.source !== "tool") return undefined; + const key = matchContext.streamKey; + if (typeof key !== "string" || !key.startsWith("toolcall:")) return undefined; + const id = key.slice("toolcall:".length); + return id.length > 0 ? id : undefined; + } + + #addPerToolInjections(toolCallId: string, rules: Rule[]): void { + const bucket = this.#perToolInjections.get(toolCallId) ?? []; + const seen = new Set(bucket.map(rule => rule.name)); + const claimedElsewhere = new Set(); + for (const [otherId, otherBucket] of this.#perToolInjections) { + if (otherId === toolCallId) continue; + for (const rule of otherBucket) claimedElsewhere.add(rule.name); + } + const newlyAdded: string[] = []; + for (const rule of rules) { + if (seen.has(rule.name) || claimedElsewhere.has(rule.name)) continue; + bucket.push(rule); + seen.add(rule.name); + newlyAdded.push(rule.name); + } + if (bucket.length === 0) return; + this.#perToolInjections.set(toolCallId, bucket); + if (newlyAdded.length > 0) this.#manager?.markInjectedByNames(newlyAdded); + } + + #markInjected(ruleNames: string[]): void { + const uniqueRuleNames = Array.from( + new Set(ruleNames.map(ruleName => ruleName.trim()).filter(ruleName => ruleName.length > 0)), + ); + if (uniqueRuleNames.length === 0) return; + this.#manager?.markInjectedByNames(uniqueRuleNames); + this.#host.sessionManager.appendTtsrInjection(uniqueRuleNames); + } + + #findAssistantIndex(targetTimestamp: number | undefined): number { + const messages = this.#host.agent.state.messages; + for (let index = messages.length - 1; index >= 0; index--) { + const message = messages[index]; + if (message.role === "assistant" && (targetTimestamp === undefined || message.timestamp === targetTimestamp)) { + return index; + } + } + return -1; + } + + #shouldInterrupt(matches: Rule[], matchContext: TtsrMatchContext): boolean { + const globalMode = this.#manager?.getSettings().interruptMode ?? "always"; + for (const rule of matches) { + const mode = rule.interruptMode ?? globalMode; + if (mode === "never") continue; + if (mode === "prose-only" && (matchContext.source === "text" || matchContext.source === "thinking")) { + return true; + } + if (mode === "tool-only" && matchContext.source === "tool") return true; + if (mode === "always") return true; + } + return false; + } + + #queueDeferredInjectionIfNeeded(message: AssistantMessage): void { + if (message.stopReason === "aborted" || message.stopReason === "error") this.#perToolInjections.clear(); + if (this.#abortPending || this.#pendingInjections.length === 0) return; + if (message.stopReason === "aborted" || message.stopReason === "error") { + this.#pendingInjections = []; + return; + } + const injection = this.#getInjectionContent(); + if (!injection) return; + this.#host.agent.followUp({ + role: "custom", + customType: "ttsr-injection", + content: injection.content, + display: false, + details: { rules: injection.rules.map(rule => rule.name) }, + attribution: "agent", + timestamp: Date.now(), + }); + this.#ensureResumePromise(); + this.#host.scheduleAgentContinue({ + delayMs: 1, + generation: this.#host.promptGeneration(), + onSkip: () => this.resolveResume(), + shouldContinue: () => { + if (this.#host.agent.state.isStreaming || !this.#host.agent.hasQueuedMessages()) { + this.resolveResume(); + return false; + } + return true; + }, + onError: () => this.resolveResume(), + }); + } + + #getStreamingToolCallBlock(message: AgentMessage, contentIndex: number): ToolCall | undefined { + if (message.role !== "assistant") return undefined; + const content = message.content; + if (!Array.isArray(content) || contentIndex < 0 || contentIndex >= content.length) return undefined; + const block = content[contentIndex]; + return block && typeof block === "object" && block.type === "toolCall" ? (block as ToolCall) : undefined; + } + + #getToolMatchContext(toolCall: ToolCall | undefined, contentIndex: number): TtsrMatchContext { + const context: TtsrMatchContext = { source: "tool" }; + if (!toolCall) return context; + context.toolName = toolCall.name; + context.streamKey = toolCall.id ? `toolcall:${toolCall.id}` : `tool:${toolCall.name}:${contentIndex}`; + context.filePaths = this.#extractToolFilePaths(toolCall); + return context; + } + + #extractToolFilePaths(toolCall: ToolCall): string[] | undefined { + const args = toolCall.arguments ?? {}; + const tool = this.#resolveTool(toolCall); + const toolPaths = tool?.matcherPaths?.(args); + if (toolPaths && toolPaths.length > 0) { + const normalized = toolPaths.flatMap(filePath => this.#normalizePathCandidates(filePath)); + if (normalized.length > 0) return Array.from(new Set(normalized)); + } + return this.#extractFilePathsFromArgs(args); + } + + #checkStream(delta: string, matchContext: TtsrMatchContext, toolCall: ToolCall | undefined): Rule[] { + if (!this.#manager) return []; + const entries = this.#resolveMatcherEntries(toolCall); + if (entries) { + const matches: Rule[] = []; + for (const entry of entries) { + matches.push(...this.#manager.checkSnapshot(entry.digest, this.#perFileContext(matchContext, entry.path))); + } + return matches; + } + const digest = this.#resolveMatcherDigest(toolCall); + return digest !== undefined + ? this.#manager.checkSnapshot(digest, matchContext) + : this.#manager.checkDelta(delta, matchContext); + } + + #resolveMatcherDigest(toolCall: ToolCall | undefined): string | undefined { + const tool = this.#resolveTool(toolCall); + return tool?.matcherDigest?.(toolCall?.arguments ?? {}); + } + + #resolveMatcherEntries(toolCall: ToolCall | undefined): readonly { path: string; digest: string }[] | undefined { + const tool = this.#resolveTool(toolCall); + const entries = tool?.matcherEntries?.(toolCall?.arguments ?? {}); + return entries && entries.length > 0 ? entries : undefined; + } + + #resolveTool(toolCall: ToolCall | undefined) { + if (!toolCall) return undefined; + const tools = this.#host.agent.state.tools; + return ( + tools.find(tool => tool.name === toolCall.name) ?? + tools.find(tool => tool.customWireName !== undefined && tool.customWireName === toolCall.name) + ); + } + + #perFileContext(base: TtsrMatchContext, filePath: string): TtsrMatchContext { + const filePaths = this.#normalizePathCandidates(filePath); + return { + ...base, + filePaths: filePaths.length > 0 ? filePaths : [filePath], + streamKey: base.streamKey ? `${base.streamKey}#${filePath}` : undefined, + }; + } + + async #checkAstStream(matchContext: TtsrMatchContext, toolCall: ToolCall | undefined): Promise { + if (!this.#manager) return []; + const entries = this.#resolveMatcherEntries(toolCall); + if (entries) { + const matches: Rule[] = []; + for (const entry of entries) { + matches.push( + ...(await this.#manager.checkAstSnapshot(entry.digest, this.#perFileContext(matchContext, entry.path))), + ); + } + return matches; + } + const digest = this.#resolveMatcherDigest(toolCall); + return digest === undefined ? [] : this.#manager.checkAstSnapshot(digest, matchContext); + } + + #handleMatches(matches: Rule[], matchContext: TtsrMatchContext, targetTimestamp: number | undefined): boolean { + const shouldInterrupt = this.#shouldInterrupt(matches, matchContext); + const matchedToolId = this.#extractToolCallId(matchContext); + const perToolId = shouldInterrupt ? undefined : matchedToolId; + if (perToolId) { + this.#addPerToolInjections(perToolId, matches); + this.#host.emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); + return false; + } + this.#addPendingInjections(matches); + if (!shouldInterrupt) return false; + + this.#abortPending = true; + this.#ensureResumePromise(); + const abortReason = this.#formatAbortReason(matches); + this.#host.agent.abort( + matchedToolId + ? createToolScopedAbortReason( + abortReason, + { [matchedToolId]: abortReason }, + "TTSR interrupt on another tool call", + ) + : abortReason, + ); + this.#host.emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); + const retryToken = ++this.#retryToken; + const generation = this.#host.promptGeneration(); + this.#host.schedulePostPromptTask( + async () => { + if (this.#retryToken !== retryToken) { + this.resolveResume(); + return; + } + const targetAssistantIndex = this.#findAssistantIndex(targetTimestamp); + if (!this.#abortPending || this.#host.promptGeneration() !== generation || targetAssistantIndex === -1) { + this.#abortPending = false; + this.#pendingInjections = []; + this.#perToolInjections.clear(); + this.resolveResume(); + return; + } + this.#abortPending = false; + this.#perToolInjections.clear(); + if (this.#manager?.getSettings().contextMode === "discard") { + this.#host.agent.replaceMessages(this.#host.agent.state.messages.slice(0, targetAssistantIndex)); + } + const injection = this.#getInjectionContent(); + if (injection) { + const details = { rules: injection.rules.map(rule => rule.name) }; + this.#host.agent.appendMessage({ + role: "custom", + customType: "ttsr-injection", + content: injection.content, + display: false, + details, + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.sessionManager.appendCustomMessageEntry( + "ttsr-injection", + injection.content, + false, + details, + "agent", + ); + this.#markInjected(details.rules); + } + try { + await this.#host.agent.continue(); + } catch { + this.resolveResume(); + } + }, + { delayMs: 50 }, + ); + return true; + } + + #extractFilePathsFromArgs(args: unknown): string[] | undefined { + if (!isRecord(args)) return undefined; + const rawPaths: string[] = []; + for (const key in args) { + const value = args[key]; + const normalizedKey = key.toLowerCase(); + if (typeof value === "string" && (normalizedKey === "path" || normalizedKey.endsWith("path"))) { + rawPaths.push(value); + continue; + } + if (Array.isArray(value) && (normalizedKey === "paths" || normalizedKey.endsWith("paths"))) { + for (const candidate of value) if (typeof candidate === "string") rawPaths.push(candidate); + } + } + const normalizedPaths = rawPaths.flatMap(filePath => this.#normalizePathCandidates(filePath)); + return normalizedPaths.length === 0 ? undefined : Array.from(new Set(normalizedPaths)); + } + + #normalizePathCandidates(rawPath: string): string[] { + const trimmed = rawPath.trim(); + if (trimmed.length === 0) return []; + const normalizedInput = trimmed.replaceAll("\\", "/"); + const candidates = new Set([normalizedInput]); + if (normalizedInput.startsWith("./")) candidates.add(normalizedInput.slice(2)); + const cwd = this.#host.sessionManager.getCwd(); + const absolutePath = path.isAbsolute(trimmed) ? path.normalize(trimmed) : path.resolve(cwd, trimmed); + candidates.add(absolutePath.replaceAll("\\", "/")); + const relative = path.relative(cwd, absolutePath).replaceAll("\\", "/"); + if (relative && relative !== "." && !relative.startsWith("../") && relative !== "..") candidates.add(relative); + return Array.from(candidates); + } +} diff --git a/packages/coding-agent/src/session/turn-recovery.ts b/packages/coding-agent/src/session/turn-recovery.ts new file mode 100644 index 000000000..5071fc7f9 --- /dev/null +++ b/packages/coding-agent/src/session/turn-recovery.ts @@ -0,0 +1,1759 @@ +import { scheduler } from "node:timers/promises"; +import { + type Agent, + AgentBusyError, + type AgentMessage, + isSyntheticToolResultMessage, + type ThinkingLevel, +} from "@oh-my-pi/pi-agent-core"; +import type { + AssistantMessage, + AssistantRetryRecovery, + AssistantRetryRecoveryKind, + CodexCompactionContext, + Model, + TextContent, + ToolChoice, +} from "@oh-my-pi/pi-ai"; +import { calculateRateLimitBackoffMs, parseRateLimitReason } from "@oh-my-pi/pi-ai"; +import * as AIError from "@oh-my-pi/pi-ai/error"; +import { kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; +import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; +import { extractRetryHint, logger, prompt } from "@oh-my-pi/pi-utils"; +import type { ModelRegistry } from "../config/model-registry"; +import { + formatModelSelectorValue, + formatModelString, + formatModelStringWithRouting, + resolveModelOverride, +} from "../config/model-resolver"; +import type { Settings } from "../config/settings"; +import type { RecoveredRetryError } from "../extensibility/shared-events"; +import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" }; +import thinkingLoopRedirectTemplate from "../prompts/system/thinking-loop-redirect.md" with { type: "text" }; +import unexpectedStopRetryTemplate from "../prompts/system/unexpected-stop-retry.md" with { type: "text" }; +import type { ConfiguredThinkingLevel } from "../thinking"; +import type { AgentSessionEvent } from "./agent-session-events"; +import type { InitialRetryFallbackState } from "./agent-session-types"; +import { isEmptyErrorTurn } from "./messages"; +import { + type ActiveRetryFallbackState, + calculateRetryBackoffDelayMs, + formatRetryFallbackBaseSelector, + formatRetryFallbackSelector, + getRetryFallbackChains, + getRetryFallbackEffectiveChain, + getRetryFallbackPrimarySelector, + getRetryFallbackRevertPolicy, + isKnownProvider, + isRetryFallbackModelKey, + isRetryFallbackWildcardKey, + parseRetryFallbackChainEntry, + parseRetryFallbackSelector, + parseRetryFallbackWildcard, + type RetryFallbackChains, + type RetryFallbackRevertPolicy, + type RetryFallbackSelector, + validateRetryFallbackChains, +} from "./retry-fallback-chains"; +import { getLatestCompactionEntry } from "./session-context"; +import { EPHEMERAL_MODEL_CHANGE_ROLE, type SessionEntry } from "./session-entries"; +import type { SessionManager } from "./session-manager"; +import { sameMessageContent, sessionMessagePersistenceKey } from "./turn-persistence"; +import { classifyUnexpectedStop, isUnexpectedStopCandidate } from "./unexpected-stop-classifier"; + +const THINKING_LOOP_REDIRECT_TYPE = "thinking-loop-redirect"; +const UNEXPECTED_STOP_MAX_RETRIES = 3; +const UNEXPECTED_STOP_TIMEOUT_MS = 4000; +const EMPTY_STOP_MAX_RETRIES = 3; +const SIBLING_UNBLOCK_BUFFER_MS = 1_000; +const NON_WHITESPACE_RE = /\S/; + +function hasNonWhitespace(value: string): boolean { + return NON_WHITESPACE_RE.test(value); +} + +function syntheticToolResultTailStart(messages: readonly AgentMessage[]): number { + let index = messages.length; + while (index > 0 && isSyntheticToolResultMessage(messages[index - 1])) { + index--; + } + return index; +} + +function retryableAssistantTurnEnd(messages: readonly AgentMessage[]): number | undefined { + const turnEnd = syntheticToolResultTailStart(messages); + const message = messages[turnEnd - 1]; + if (message?.role !== "assistant") return undefined; + if (message.stopReason !== "error" && message.stopReason !== "aborted") return undefined; + return turnEnd; +} + +/** Result shape shared with automatic maintenance recovery. */ +export interface RecoveryCompactionResult { + deferredHandoff: boolean; + continuationScheduled: boolean; + automaticContinuationBlocked?: boolean; + historyRewritten?: boolean; +} + +/** Capabilities borrowed from the owning AgentSession. */ +export interface TurnRecoveryHost { + agent: Agent; + sessionManager: SessionManager; + settings: Settings; + modelRegistry: ModelRegistry; + configWarnings: string[]; + model(): Model | undefined; + thinkingLevel(): ThinkingLevel | undefined; + configuredThinkingLevel(): ConfiguredThinkingLevel | undefined; + setThinkingLevel(level: ConfiguredThinkingLevel | undefined): void; + isDisposed(): boolean; + isStreaming(): boolean; + isCompacting(): boolean; + abortInProgress(): boolean; + streamingEditAbortTriggered(): boolean; + promptGeneration(): number; + sessionId(): string; + emitSessionEvent(event: AgentSessionEvent): Promise; + scheduleAgentContinue(options: { delayMs?: number; generation?: number }): void; + waitForSessionMessagePersistence(message: AssistantMessage): Promise; + appendSessionMessage(message: AssistantMessage): void; + sessionMessageAlreadyPersisted(message: AssistantMessage): boolean; + setModelWithProviderSessionReset(model: Model): void; + resetCurrentResponsesProviderSession(reason: string): void; + maybeAutoRedeemCodexReset(): Promise; + runAutoCompaction( + reason: "overflow" | "threshold" | "idle" | "incomplete", + willRetry: boolean, + deferred?: boolean, + allowDefer?: boolean, + options?: { + autoContinue?: boolean; + triggerContextTokens?: number; + suppressContinuation?: boolean; + suppressHandoff?: boolean; + phase?: CodexCompactionContext["phase"]; + terminalTextAnswer?: boolean; + }, + ): Promise; + withBashBranchTransition(operation: () => T): T; +} + +/** Construction-time retry state restored from model selection. */ +export interface TurnRecoveryOptions { + initialRetryFallback?: InitialRetryFallbackState; +} + +type PendingRecoveredRetryError = { + entryId: string; + persistenceKey: string; + recovery: AssistantRetryRecoveryKind; + attempt: number; + note: string; +}; + +/** Owns terminal-stop recovery, automatic retries, and fallback routing. */ +export class TurnRecovery { + readonly #host: TurnRecoveryHost; + #retryAbortController: AbortController | undefined; + #retryAttempt = 0; + #retryPromise: Promise | undefined; + #retryResolve: (() => void) | undefined; + #activeRetryFallback: ActiveRetryFallbackState | undefined; + #pendingRecoveredRetryErrors: PendingRecoveredRetryError[] = []; + #emptyStopRetryCount = 0; + #unexpectedStopRetryCount = 0; + #acceptTerminalEmptyStopForPrompt = false; + + constructor(host: TurnRecoveryHost, options: TurnRecoveryOptions = {}) { + this.#host = host; + if (options.initialRetryFallback) { + this.#activeRetryFallback = { + ...options.initialRetryFallback, + lastAppliedFallbackThinkingLevel: host.configuredThinkingLevel(), + pinned: false, + }; + } + this.#validateRetryFallbackChains(); + } + + /** Current automatic retry attempt. */ + get attempt(): number { + return this.#retryAttempt; + } + + /** Promise settled when the active retry saga finishes. */ + get retryPromise(): Promise | undefined { + return this.#retryPromise; + } + + /** Resolved selector while fallback routing owns the current model. */ + get retryFallbackModel(): string | undefined { + const model = this.#host.model(); + return this.#activeRetryFallback && model + ? formatRetryFallbackSelector(model, this.#host.thinkingLevel()) + : undefined; + } + + /** Resets per-prompt recovery counters and terminal-stop acceptance. */ + resetForNewPrompt(): void { + this.#emptyStopRetryCount = 0; + this.#unexpectedStopRetryCount = 0; + this.#acceptTerminalEmptyStopForPrompt = false; + } + + /** Sets whether one terminal empty stop is accepted for the current prompt. */ + setAcceptTerminalEmptyStop(accept: boolean): void { + this.#acceptTerminalEmptyStopForPrompt = accept; + } + + /** Closes a successful retry saga and annotates recovered persisted errors. */ + async onAssistantSettledSuccessfully(message: AssistantMessage): Promise { + if ( + message.stopReason === "error" || + message.stopReason === "aborted" || + this.#isEmptyAssistantStop(message) || + this.#retryAttempt === 0 + ) { + return; + } + const model = this.#host.model(); + if (this.#activeRetryFallback && model) { + await this.#host.emitSessionEvent({ + type: "retry_fallback_succeeded", + model: formatRetryFallbackSelector(model, this.#host.thinkingLevel()), + role: this.#activeRetryFallback.role, + }); + } + const recoveredErrors = await this.#markPendingRecoveredRetryErrors(message); + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: true, + attempt: this.#retryAttempt, + recoveredErrors, + }); + this.#clearPendingRecoveredRetryErrors(); + this.#retryAttempt = 0; + } + + /** Closes a failed retry saga when no compaction continuation took ownership. */ + async onErrorSettledWithoutRetry(message: AssistantMessage, compaction: RecoveryCompactionResult): Promise { + if (message.stopReason !== "error" || this.#retryAttempt === 0 || compaction.continuationScheduled) return; + const attempt = this.#retryAttempt; + this.#retryAttempt = 0; + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt, + finalError: message.errorMessage, + }); + this.#clearPendingRecoveredRetryErrors(); + } + + /** Persists an otherwise skipped terminal empty error turn. */ + persistTerminalEmptyErrorTurn(message: AssistantMessage): Promise { + return this.#persistTerminalEmptyErrorTurn(message); + } + + /** Handles empty terminal assistant turns and schedules bounded recovery. */ + handleEmptyAssistantStop(message: AssistantMessage): Promise { + return this.#handleEmptyAssistantStop(message); + } + + /** Classifies suspicious terminal stops and schedules bounded recovery. */ + handleUnexpectedAssistantStop(message: AssistantMessage): Promise { + return this.#handleUnexpectedAssistantStop(message); + } + + /** Removes a persisted failed assistant turn after its persistence slot settles. */ + dropPersistedAssistantTurn(message: AssistantMessage): Promise { + return this.#dropPersistedAssistantTurn(message); + } + + /** Runs recovery compaction and restores the failed turn when no rewrite occurs. */ + runRecoveryCompactionWithRollback( + reason: "overflow" | "incomplete", + message: AssistantMessage, + allowDefer: boolean, + options: { autoContinue: boolean; triggerContextTokens?: number }, + ): Promise { + return this.#runRecoveryCompactionWithRollback(reason, message, allowDefer, options); + } + + /** Restores the configured primary after fallback cooldown expiry. */ + maybeRestoreRetryFallbackPrimary(): Promise { + return this.#maybeRestoreRetryFallbackPrimary(); + } + + /** Applies automatic retry, credential rotation, and model fallback policy. */ + handleRetryableError( + message: AssistantMessage, + options?: { + allowModelFallback?: boolean; + fireworksFastFallback?: boolean; + hardErrorFallback?: boolean; + preserveFailedTurn?: boolean; + }, + ): Promise { + return this.#handleRetryableError(message, options); + } + + /** Prompts after transient overlap with a prior agent run. */ + promptAgentWithIdleRetry(messages: AgentMessage[], options?: { toolChoice?: ToolChoice }): Promise { + return this.#promptAgentWithIdleRetry(messages, options); + } + + /** Parses provider retry and rate-limit reset hints into a delay. */ + parseRetryAfterMsFromError(errorMessage: string): number | undefined { + return this.#parseRetryAfterMsFromError(errorMessage); + } + + /** Resolve the pending retry promise */ + resolveRetry(): void { + if (this.#retryResolve) { + this.#retryResolve(); + this.#retryResolve = undefined; + this.#retryPromise = undefined; + } + } + + #clearPendingRecoveredRetryErrors(): void { + this.#pendingRecoveredRetryErrors = []; + } + + /** + * Durably record a terminal empty error turn (`stopReason: "error"` with no + * substantive content) that `#persistSessionMessageIfMissing` skipped, so the + * session JSONL keeps a record of why the run stopped instead of ending at the + * last tool result. A no-op for non-empty/non-error turns and idempotent via + * the already-persisted guard; the turn is dropped from active context by the + * caller (or `isProviderRefusalMessage`/`isEmptyErrorTurn` filters) so it is + * never replayed on the wire. Used by the retry-lifecycle dead-ends and the + * non-retry terminal error tail. + */ + async #persistTerminalEmptyErrorTurn(message: AssistantMessage): Promise { + await this.#host.waitForSessionMessagePersistence(message); + if (!isEmptyErrorTurn(message)) return; + if (this.#host.sessionMessageAlreadyPersisted(message)) return; + this.#host.appendSessionMessage(message); + } + + #retryRecoveryKind( + id: number, + switchedCredential: boolean, + switchedModel: boolean, + delayMs: number, + ): AssistantRetryRecoveryKind { + if (switchedCredential) return "credential"; + if (switchedModel) return "model"; + if (AIError.is(id, AIError.Flag.UsageLimit) && delayMs > 0) return "wait"; + return "plain"; + } + + #retryRecoveryNote(recovery: AssistantRetryRecoveryKind, rateLimited: boolean): string { + const parts: string[] = []; + if (rateLimited) { + parts.push("rate-limited"); + } else if (recovery === "plain") { + parts.push("error"); + } + if (recovery === "credential") { + parts.push("switched account"); + } else if (recovery === "model") { + parts.push("switched model"); + } else if (recovery === "wait") { + parts.push("waited"); + } + parts.push("retried"); + return parts.join("; "); + } + + async #recordPendingRecoveredRetryError( + message: AssistantMessage, + id: number, + options: { switchedCredential: boolean; switchedModel: boolean; delayMs: number }, + ): Promise { + await this.persistTerminalEmptyErrorTurn(message); + const persistenceKey = sessionMessagePersistenceKey(message); + if (!persistenceKey) return; + let branchEntry: SessionEntry | undefined; + for (const entry of this.#host.sessionManager.getBranch().slice().reverse()) { + if (entry.type !== "message" || entry.message.role !== "assistant") continue; + if (sessionMessagePersistenceKey(entry.message) !== persistenceKey) continue; + if (!sameMessageContent(entry.message, message) && !this.#isSameAssistantMessage(entry.message, message)) { + continue; + } + branchEntry = entry; + break; + } + if (!branchEntry) return; + if (this.#pendingRecoveredRetryErrors.some(error => error.entryId === branchEntry.id)) return; + const rateLimited = AIError.is(id, AIError.Flag.UsageLimit); + const recovery = this.#retryRecoveryKind(id, options.switchedCredential, options.switchedModel, options.delayMs); + const note = this.#retryRecoveryNote(recovery, rateLimited); + this.#pendingRecoveredRetryErrors.push({ + entryId: branchEntry.id, + persistenceKey, + recovery, + attempt: this.#retryAttempt, + note, + }); + } + + async #markPendingRecoveredRetryErrors(supersedingMessage: AssistantMessage): Promise { + if (this.#pendingRecoveredRetryErrors.length === 0) return []; + const branch = this.#host.sessionManager.getBranch(); + const branchById = new Map(); + for (const entry of branch) { + branchById.set(entry.id, entry); + } + const recoveredAt = new Date().toISOString(); + const supersededBy: AssistantRetryRecovery["supersededBy"] = { + timestamp: supersedingMessage.timestamp, + provider: supersedingMessage.provider, + model: supersedingMessage.model, + }; + if (supersedingMessage.responseId) { + supersededBy.responseId = supersedingMessage.responseId; + } + const recoveredErrors: RecoveredRetryError[] = []; + for (const pending of this.#pendingRecoveredRetryErrors) { + let entry = branchById.get(pending.entryId); + if (entry?.type !== "message" || entry.message.role !== "assistant") { + entry = branch + .slice() + .reverse() + .find( + candidate => + candidate.type === "message" && + candidate.message.role === "assistant" && + sessionMessagePersistenceKey(candidate.message) === pending.persistenceKey, + ); + } + if (entry?.type !== "message" || entry.message.role !== "assistant") continue; + const retryRecovery: AssistantRetryRecovery = { + kind: "auto-retry", + status: "recovered", + attempt: pending.attempt, + recoveredAt, + recovery: pending.recovery, + note: pending.note, + supersededBy, + }; + entry.message.retryRecovery = retryRecovery; + recoveredErrors.push({ + entryId: entry.id, + persistenceKey: pending.persistenceKey, + note: retryRecovery.note, + retryRecovery, + }); + } + if (recoveredErrors.length > 0) { + await this.#host.sessionManager.rewriteEntries(); + } + return recoveredErrors; + } + + async #handleEmptyAssistantStop(assistantMessage: AssistantMessage): Promise { + if (!this.#isEmptyAssistantStop(assistantMessage)) { + this.#emptyStopRetryCount = 0; + return false; + } + + if (this.#acceptTerminalEmptyStopForPrompt && assistantMessage.stopReason === "stop") { + this.#acceptTerminalEmptyStopForPrompt = false; + this.#discardAcceptedTerminalEmptyStop(assistantMessage); + this.#emptyStopRetryCount = 0; + return false; + } + + this.#emptyStopRetryCount++; + if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) { + const attempts = this.#emptyStopRetryCount - 1; + const finalError = + "Assistant returned empty stop after retry cap; try switching models or `/shake images` to remove archived frames"; + logger.warn(finalError, { + attempts, + model: assistantMessage.model, + provider: assistantMessage.provider, + }); + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt: this.#retryAttempt > 0 ? this.#retryAttempt : attempts, + finalError, + }); + this.#clearPendingRecoveredRetryErrors(); + this.#retryAttempt = 0; + this.resolveRetry(); + // A zero-content turn carries no transcript value, while its provider usage + // can anchor the next prompt at the full failed-request size and re-trigger + // compaction at the same boundary. Remove every capped empty stop; toolUse + // orphans still need this for Anthropic message-history validity. + await this.dropPersistedAssistantTurn(assistantMessage); + return false; + } + this.discardAssistantTurn(assistantMessage); + this.#host.agent.appendMessage({ + role: "developer", + content: [{ type: "text", text: this.#emptyStopRetryReminder() }], + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.scheduleAgentContinue({ generation: this.#host.promptGeneration() }); + return true; + } + + #isEmptyAssistantStop(assistantMessage: AssistantMessage): boolean { + switch (assistantMessage.stopReason) { + case "stop": + // Unsigned thinking alone is not actionable, but a signature is + // provider-authenticated content and makes the stop terminal. + for (const content of assistantMessage.content) { + if (content.type === "toolCall") return false; + if (content.type === "text" && hasNonWhitespace(content.text)) return false; + if (content.type === "thinking" && hasNonWhitespace(content.thinkingSignature ?? "")) return false; + } + return true; + case "toolUse": + // An orphaned toolUse stop (no tool_use block) corrupts Anthropic history: + // a later tool_result has nothing to anchor to. Thinking alone cannot anchor + // a tool_result, so it does not rescue a toolUse stop here. + for (const content of assistantMessage.content) { + if (content.type === "toolCall") return false; + if (content.type === "text" && hasNonWhitespace(content.text)) return false; + } + return true; + default: + return false; + } + } + + #emptyStopRetryReminder(): string { + return prompt.render(emptyStopRetryTemplate, { + retryCount: this.#emptyStopRetryCount, + maxRetries: EMPTY_STOP_MAX_RETRIES, + }); + } + async #handleUnexpectedAssistantStop(assistantMessage: AssistantMessage): Promise { + if (!this.#host.settings.get("features.unexpectedStopDetection")) { + return false; + } + if (!isUnexpectedStopCandidate(assistantMessage)) { + this.#unexpectedStopRetryCount = 0; + return false; + } + + const text = assistantMessage.content + .filter((content): content is TextContent => content.type === "text") + .map(content => content.text) + .join("\n"); + if (!/\S/.test(text)) { + this.#unexpectedStopRetryCount = 0; + return false; + } + + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), UNEXPECTED_STOP_TIMEOUT_MS); + let classification: boolean | undefined; + try { + classification = await classifyUnexpectedStop(text, { + settings: this.#host.settings, + registry: this.#host.modelRegistry, + sessionId: this.#host.sessionId(), + metadataResolver: (provider: string) => this.#host.agent.metadataForProvider(provider), + signal: controller.signal, + }); + } finally { + clearTimeout(timeout); + } + + if (classification !== true) { + this.#unexpectedStopRetryCount = 0; + return false; + } + + this.#unexpectedStopRetryCount++; + if (this.#unexpectedStopRetryCount > UNEXPECTED_STOP_MAX_RETRIES) { + logger.warn("Assistant returned unexpected stop after retry cap", { + attempts: this.#unexpectedStopRetryCount - 1, + model: assistantMessage.model, + provider: assistantMessage.provider, + }); + this.#unexpectedStopRetryCount = 0; + return false; + } + + this.#host.agent.appendMessage({ + role: "developer", + content: [{ type: "text", text: this.#unexpectedStopRetryReminder() }], + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.scheduleAgentContinue({ generation: this.#host.promptGeneration() }); + return true; + } + + #unexpectedStopRetryReminder(): string { + return prompt.render(unexpectedStopRetryTemplate, { + retryCount: this.#unexpectedStopRetryCount, + maxRetries: UNEXPECTED_STOP_MAX_RETRIES, + }); + } + + removeAssistantMessageFromActiveContext( + assistantMessage: AssistantMessage, + reason = "assistant-context-cleanup", + ): void { + const messages = this.#host.agent.state.messages; + const lastMessage = messages[messages.length - 1]; + const lastAssistant: AssistantMessage | undefined = lastMessage?.role === "assistant" ? lastMessage : undefined; + if (lastAssistant !== undefined && this.#isSameAssistantMessage(lastAssistant, assistantMessage)) { + this.#host.agent.replaceMessages(messages.slice(0, -1)); + return; + } + // A miss means the failed turn is still in active context (or was never + // there); log just enough to explain why the identity check failed. + logger.debug("agent active context assistant removal missed", { + reason, + lastRole: lastMessage?.role, + candidateTimestamp: assistantMessage.timestamp, + lastTimestamp: lastAssistant?.timestamp, + candidateStopReason: assistantMessage.stopReason, + lastStopReason: lastAssistant?.stopReason, + }); + } + + /** + * Drop a recoverable assistant turn from the persisted session branch once a + * recovery path (context promotion or compaction) is committed. Waits for the + * in-flight `message_end` persistence slot first so the branch entry exists + * before we reparent past it. Active context removal is the caller's + * responsibility — recovery paths clear it eagerly so the retry never + * replays the failed turn, while no-recovery paths leave the persisted entry + * (and the user-visible transcript line) in place. + */ + async #dropPersistedAssistantTurn(assistantMessage: AssistantMessage): Promise { + await this.#host.waitForSessionMessagePersistence(assistantMessage); + this.discardAssistantTurn(assistantMessage); + } + + /** + * Drop the failed assistant turn from persisted history, run + * {@link #runAutoCompaction} for an `overflow` / `incomplete` recovery, and + * restore the assistant entry if compaction did not actually commit + * anything (no usable model/preparation, hook cancel, compaction error, + * or a no-progress automatic-continuation block before any summary was + * written). + * + * Compaction has to see a clean branch — otherwise its `prepareCompaction` + * pass would keep the failed turn in the kept region and the retry would + * replay it. But a return that was not paired with a fresh compaction + * summary or a successful history rewrite means no recovery is in progress, + * even if queued user input gets drained next. Restoring the failed turn + * before that continuation preserves the visible stop reason and rebuilds the + * active assistant tail that `Agent.continue()` needs to dequeue follow-ups. + */ + async #runRecoveryCompactionWithRollback( + reason: "overflow" | "incomplete", + assistantMessage: AssistantMessage, + allowDefer: boolean, + options: { autoContinue: boolean; triggerContextTokens?: number }, + ): Promise { + const compactionEntryBefore = getLatestCompactionEntry(this.#host.sessionManager.getBranch()); + await this.dropPersistedAssistantTurn(assistantMessage); + const result = await this.#host.runAutoCompaction(reason, true, false, allowDefer, { + autoContinue: options.autoContinue, + triggerContextTokens: options.triggerContextTokens, + phase: "mid_turn", + }); + const compactionEntryAfter = getLatestCompactionEntry(this.#host.sessionManager.getBranch()); + if (result.historyRewritten !== true && compactionEntryAfter === compactionEntryBefore) { + this.#restoreFailedAssistantTurn(assistantMessage); + } + return result; + } + + #restoreFailedAssistantTurn(assistantMessage: AssistantMessage): void { + if (!isEmptyErrorTurn(assistantMessage)) this.#host.sessionManager.appendMessage(assistantMessage); + const lastMessage = this.#host.agent.state.messages.at(-1); + if ( + lastMessage?.role === "assistant" && + this.#isSameAssistantMessage(lastMessage as AssistantMessage, assistantMessage) + ) { + return; + } + this.#host.agent.appendMessage(assistantMessage); + } + + #discardAcceptedTerminalEmptyStop(assistantMessage: AssistantMessage): void { + const branch = this.#host.sessionManager.getBranch(); + const branchEntry = branch + .slice() + .reverse() + .find( + entry => + entry.type === "message" && + entry.message.role === "assistant" && + this.#isSameAssistantMessage(entry.message, assistantMessage), + ); + const parentEntry = + branchEntry?.parentId === null || branchEntry?.parentId === undefined + ? undefined + : branch.find(entry => entry.id === branchEntry.parentId); + const prunePrompt = parentEntry?.type === "custom_message"; + + this.removeAssistantMessageFromActiveContext(assistantMessage, "accepted-terminal-empty-stop"); + if (prunePrompt && this.#host.agent.state.messages.at(-1)?.role === "custom") { + this.#host.agent.replaceMessages(this.#host.agent.state.messages.slice(0, -1)); + } + + if (!branchEntry) return; + const targetParentId = prunePrompt ? parentEntry.parentId : branchEntry.parentId; + this.#host.withBashBranchTransition(() => { + if (targetParentId === null) { + this.#host.sessionManager.resetLeaf(); + } else { + this.#host.sessionManager.branch(targetParentId); + } + }); + this.#host.sessionManager.appendCustomEntry("accepted-terminal-empty-stop"); + } + + /** + * Drop an assistant turn from BOTH the live agent context and the persisted + * session branch (reparenting the leaf to the turn's parent), so a discarded + * turn does not resurface on reload. Used for empty/reasoning-only stops and + * the Gemini header-runaway interrupt, which must not replay a partial, + * loop-fueling thinking block. + */ + discardAssistantTurn(assistantMessage: AssistantMessage): void { + this.removeAssistantMessageFromActiveContext(assistantMessage); + + const branchEntry = this.#host.sessionManager + .getBranch() + .slice() + .reverse() + .find( + entry => + entry.type === "message" && + entry.message.role === "assistant" && + this.#isSameAssistantMessage(entry.message as AssistantMessage, assistantMessage), + ); + if (!branchEntry) { + return; + } + this.#host.withBashBranchTransition(() => { + if (branchEntry.parentId === null) { + this.#host.sessionManager.resetLeaf(); + } else { + this.#host.sessionManager.branch(branchEntry.parentId); + } + }); + } + + #isSameAssistantMessage(left: AssistantMessage, right: AssistantMessage): boolean { + return ( + left === right || + (left.timestamp === right.timestamp && + left.provider === right.provider && + left.model === right.model && + left.stopReason === right.stopReason) + ); + } + + /** + * Classify retry decisions against the active session model. Test stream + * shims and provider adapters can emit generic assistant metadata, but retry + * policy belongs to the model that was actually requested for this turn. + */ + #classifyRetryMessage(message: AssistantMessage): number { + const activeModel = this.#host.model(); + if (!activeModel || message.api === activeModel.api) { + return AIError.classifyMessage(message); + } + + const id = AIError.classifyMessage({ + api: activeModel.api, + errorId: message.errorId, + errorMessage: message.errorMessage, + errorStatus: message.errorStatus, + }); + message.errorId = id; + return id; + } + + #isGenericAbortSentinel(message: AssistantMessage): boolean { + return message.errorMessage === "Request was aborted" || message.errorMessage === "Request was aborted."; + } + + /** + * Retry an empty, reason-less provider abort: a turn with no content that + * carries the generic sentinel (bare `abort()`), whether the provider + * finalized it as `stopReason: "aborted"` or leaked it as `stopReason: + * "error"` (a stalled/dropped stream reported as an error rather than an + * abort — issue #5375). Only fires while the session is neither aborting nor + * tearing down. A user/lifecycle abort (`#abortInProgress`), a dispose-driven + * abort (`#isDisposed`), or a session-induced streaming-edit guard abort + * (`StreamingEditGuard.abortTriggered` — auto-generated-file guard or failed-patch + * preview) is deliberate and MUST settle the turn instead: routing it through + * retry would orphan `#retryPromise` on a continuation the guard skips + * (hanging the in-flight `prompt()`) or silently undo the guard's intended + * abort. Deliberate user interrupts (`UserInterrupt`) and silent aborts carry + * their own marker, not the generic sentinel, so they never match here. + */ + isRetryableReasonlessAbort(message: AssistantMessage): boolean { + if ( + (message.stopReason !== "aborted" && message.stopReason !== "error") || + message.content.length !== 0 || + this.#host.abortInProgress() || + this.#host.isDisposed() || + this.#host.streamingEditAbortTriggered() + ) { + return false; + } + + const id = this.#classifyRetryMessage(message); + if (message.stopReason === "aborted" && AIError.is(id, AIError.Flag.Abort)) return true; + if (!this.#isGenericAbortSentinel(message)) return false; + + message.errorId = AIError.create(AIError.Flag.Abort); + return true; + } + + /** + * Check if an error is retryable (transient errors or usage limits). + * Context overflow is NOT retryable (handled by compaction instead). + * Usage-limit errors are retryable because the retry handler performs credential switching. + */ + isRetryableError(message: AssistantMessage): boolean { + if (message.stopReason !== "error") return false; + + const id = this.#classifyRetryMessage(message); + // Context overflow is handled by compaction, not retry + const contextWindow = this.#host.model()?.contextWindow ?? 0; + if (AIError.isContextOverflow(message, contextWindow)) return false; + + if (this.isClassifierRefusal(message)) return true; + return AIError.retriable(id, { replayUnsafe: this.#hasReplayUnsafeToolOutput(message) }); + } + + /** + * Resume a stalled turn after every emitted tool call has produced a result. + * Cursor calls must also carry the server-execution marker. The failed + * assistant/tool-result pair stays in context so completed side effects are + * continued from rather than replayed. + */ + canResumeResolvedStreamStall(message: AssistantMessage): boolean { + if (message.stopReason !== "error" || !message.errorMessage?.toLowerCase().includes("stream stall")) { + return false; + } + const id = this.#classifyRetryMessage(message); + if (!AIError.retriable(id)) return false; + + const resolvedToolCallIds: string[] = []; + for (const block of message.content) { + if (block.type !== "toolCall") continue; + if ( + message.provider === "cursor" && + (!(kCursorExecResolved in block) || block[kCursorExecResolved] !== true) + ) { + return false; + } + resolvedToolCallIds.push(block.id); + } + if (resolvedToolCallIds.length === 0) return false; + + const messages = this.#host.agent.state.messages; + let assistantIndex = -1; + for (let i = messages.length - 1; i >= 0; i--) { + const candidate = messages[i]; + if (candidate.role === "assistant" && this.#isSameAssistantMessage(candidate, message)) { + assistantIndex = i; + break; + } + } + if (assistantIndex < 0) return false; + + const unresolvedToolCallIds = new Set(resolvedToolCallIds); + for (let i = assistantIndex + 1; i < messages.length; i++) { + const candidate = messages[i]; + if (candidate.role === "toolResult") unresolvedToolCallIds.delete(candidate.toolCallId); + } + return unresolvedToolCallIds.size === 0; + } + /** + * Retried turns remove the failed assistant message from active context. + * Text/thinking-only partials are safe to discard and replay. Retained + * tool calls are not: a completed tool call may already have emitted its + * tool result after this assistant message, so replaying can duplicate work. + */ + #hasReplayUnsafeToolOutput(message: AssistantMessage): boolean { + return message.content.some(block => block.type === "toolCall"); + } + + /** + * OpenRouter can repeatedly close Gemini streams at the reasoning-to-payload + * transition. One retry covers a transient edge failure; the normal ten-retry + * budget would otherwise re-run the same expensive reasoning cycle unchanged. + */ + #isOpenRouterThinkingStreamClose(message: AssistantMessage): boolean { + return ( + message.provider === "openrouter" && + /server_error:\s*stream closed with reason:\s*error/i.test(message.errorMessage ?? "") && + message.content.some(block => block.type === "thinking" && block.thinking.trim().length > 0) + ); + } + + /** Checks whether a provider error represents a classifier refusal. */ + isClassifierRefusal(message: AssistantMessage): boolean { + if (message.stopReason !== "error") return false; + const stopType = message.stopDetails?.type; + return stopType === "refusal" || stopType === "sensitive"; + } + + /** + * True when `provider` has registered models or is configured for dynamic + * discovery. Discovery-only providers (e.g. a models.yml provider with + * `discovery:` and no static models) can hold zero models until the online + * refresh completes, so a models-only check would misreport them as + * unknown during session construction. + */ + #isKnownProvider(provider: string): boolean { + return isKnownProvider(this.#host.modelRegistry, provider); + } + + #getRetryFallbackChains(): RetryFallbackChains { + return getRetryFallbackChains(this.#host.settings); + } + + #validateRetryFallbackChains(): void { + validateRetryFallbackChains(this.#host.settings, this.#host.modelRegistry, message => + this.#host.configWarnings.push(message), + ); + } + + #getRetryFallbackRevertPolicy(): RetryFallbackRevertPolicy { + return getRetryFallbackRevertPolicy(this.#host.settings); + } + + #getRetryFallbackPrimarySelector(role: string): RetryFallbackSelector | undefined { + return getRetryFallbackPrimarySelector(this.#host.settings, this.#host.modelRegistry, role); + } + + /** Clears fallback ownership after an explicit model change. */ + clearActiveRetryFallback(): void { + this.#activeRetryFallback = undefined; + } + + /** Checks whether a fallback selector remains in cooldown. */ + isRetryFallbackSelectorSuppressed(selector: RetryFallbackSelector): boolean { + return this.#host.modelRegistry.isSelectorSuppressed(selector.raw); + } + + /** Records the cooldown that should suppress a failing selector. */ + noteRetryFallbackCooldown(currentSelector: string, retryAfterMs: number | undefined, errorMessage: string): void { + let cooldownMs = retryAfterMs; + if (!cooldownMs || cooldownMs <= 0) { + const reason = parseRateLimitReason(errorMessage); + cooldownMs = reason === "UNKNOWN" ? 5 * 60 * 1000 : calculateRateLimitBackoffMs(reason); + } + this.#host.modelRegistry.suppressSelector(currentSelector, Date.now() + cooldownMs); + } + + /** + * Map the failing model selector to the chain key that owns it, by + * specificity: an exact model-selector key, then a `provider/*` wildcard, + * then a model role whose current assignment matches, then `default`. + * Model-oriented keys win over roles so a chain follows the model across + * role reassignments. + */ + resolveRetryFallbackRole( + currentSelector: string, + currentModel: Model | null | undefined = this.#host.model(), + ): string | undefined { + const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#host.modelRegistry); + if (!parsedCurrent) return undefined; + const chains = this.#getRetryFallbackChains(); + const currentBaseSelector = formatRetryFallbackBaseSelector(parsedCurrent); + const currentPlainSelector = currentModel + ? formatModelSelectorValue(formatModelString(currentModel), parsedCurrent.thinkingLevel) + : undefined; + const currentPlainBaseSelector = + currentPlainSelector && currentPlainSelector !== currentSelector + ? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent) + : undefined; + + const exactModelKeys: string[] = []; + const roleKeys: string[] = []; + for (const key in chains) { + if (!isRetryFallbackModelKey(key)) roleKeys.push(key); + else if (!isRetryFallbackWildcardKey(key)) exactModelKeys.push(key); + } + const matchesCurrent = (primary: RetryFallbackSelector | undefined): boolean => { + if (!primary) return false; + if (primary.raw === currentSelector || (currentPlainSelector && primary.raw === currentPlainSelector)) { + return true; + } + const base = formatRetryFallbackBaseSelector(primary); + return base === currentBaseSelector || (!!currentPlainBaseSelector && base === currentPlainBaseSelector); + }; + + // 1. Exact model-selector keys — most specific. + for (const key of exactModelKeys) { + if (matchesCurrent(this.#getRetryFallbackPrimarySelector(key))) return key; + } + // 2. Provider wildcards — an id-prefixed key (`openrouter/google/*`) + // beats the plain `provider/*` key for ids under its prefix. + let wildcardMatch: string | undefined; + let wildcardPrefixLength = -1; + for (const key in chains) { + if (!isRetryFallbackWildcardKey(key) || !Array.isArray(chains[key])) continue; + const { provider, idPrefix } = parseRetryFallbackWildcard(key, p => this.#isKnownProvider(p)); + if (provider !== parsedCurrent.provider) continue; + if (idPrefix !== undefined && !parsedCurrent.id.startsWith(`${idPrefix}/`)) continue; + const prefixLength = idPrefix === undefined ? 0 : idPrefix.length; + if (prefixLength > wildcardPrefixLength) { + wildcardMatch = key; + wildcardPrefixLength = prefixLength; + } + } + if (wildcardMatch) return wildcardMatch; + // 3. Role keys — matched by the role's currently-assigned model. + for (const key of roleKeys) { + if (matchesCurrent(this.#getRetryFallbackPrimarySelector(key))) return key; + } + // 4. The default chain, when default has no explicit role primary. + const defaultChain = chains.default; + if ( + Array.isArray(defaultChain) && + defaultChain.length > 0 && + this.#getRetryFallbackPrimarySelector("default") === undefined + ) { + return "default"; + } + return undefined; + } + + /** + * Parse one configured chain entry. A `provider/*` entry keeps the failing + * model's id and swaps the provider (google-antigravity/x → google/x); an + * id-prefixed `provider/prefix/*` entry re-prefixes the failing model's + * bare id instead (openrouter/google/* : google-antigravity/x → + * openrouter/google/x). Ids the target provider lacks are skipped by the + * candidate loop's registry lookup. + */ + #parseRetryFallbackChainEntry( + entry: string, + current: RetryFallbackSelector | undefined, + ): RetryFallbackSelector | undefined { + return parseRetryFallbackChainEntry(entry, current, this.#host.modelRegistry); + } + + #getRetryFallbackEffectiveChain(role: string, currentSelector?: string): RetryFallbackSelector[] { + return getRetryFallbackEffectiveChain(this.#host.settings, this.#host.modelRegistry, role, currentSelector); + } + + /** Finds fallback candidates that follow the active selector. */ + findRetryFallbackCandidates( + role: string, + currentSelector: string, + currentModel: Model | null | undefined = this.#host.model(), + ): RetryFallbackSelector[] { + let chain = this.#getRetryFallbackEffectiveChain(role, currentSelector); + const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#host.modelRegistry); + if (chain.length === 0 && role === "default" && parsedCurrent) { + const chains = this.#getRetryFallbackChains(); + const defaultChain = chains.default; + if ( + Array.isArray(defaultChain) && + defaultChain.length > 0 && + this.#getRetryFallbackPrimarySelector("default") === undefined + ) { + const seen = new Set([parsedCurrent.raw]); + chain = [parsedCurrent]; + for (const selector of defaultChain) { + const parsed = this.#parseRetryFallbackChainEntry(selector, parsedCurrent); + if (!parsed || seen.has(parsed.raw)) continue; + seen.add(parsed.raw); + chain.push(parsed); + } + } + } + if (chain.length <= 1) return []; + const currentBaseSelector = parsedCurrent ? formatRetryFallbackBaseSelector(parsedCurrent) : undefined; + const currentPlainSelector = + currentModel && parsedCurrent + ? formatModelSelectorValue(formatModelString(currentModel), parsedCurrent.thinkingLevel) + : undefined; + const currentPlainBaseSelector = + parsedCurrent && currentPlainSelector && currentPlainSelector !== currentSelector + ? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent) + : undefined; + const exactIndex = chain.findIndex( + selector => selector.raw === currentSelector || selector.raw === currentPlainSelector, + ); + if (exactIndex >= 0) return chain.slice(exactIndex + 1); + const baseIndex = currentBaseSelector + ? chain.findIndex(selector => { + const selectorBase = formatRetryFallbackBaseSelector(selector); + return selectorBase === currentBaseSelector || selectorBase === currentPlainBaseSelector; + }) + : -1; + if (baseIndex >= 0) return chain.slice(baseIndex + 1); + return chain.slice(1); + } + + async #applyRetryFallbackCandidate( + role: string, + selector: RetryFallbackSelector, + currentSelector: string, + options?: { pinFallback?: boolean }, + ): Promise { + const resolved = resolveModelOverride([selector.raw], this.#host.modelRegistry, this.#host.settings); + const candidate = resolved.model ?? this.#host.modelRegistry.find(selector.provider, selector.id); + if (!candidate) { + throw new Error(`Retry fallback model not found: ${selector.raw}`); + } + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId()); + if (!apiKey) { + throw new Error(`No API key for retry fallback ${selector.raw}`); + } + + // Capture the configured selector (auto-aware) so a fallback chain preserves + // `auto` instead of collapsing it to the level it resolved to this turn. + const currentThinkingLevel = this.#host.configuredThinkingLevel(); + const nextThinkingLevel = selector.thinkingLevel ?? currentThinkingLevel; + const candidateSelector = formatModelStringWithRouting(candidate); + this.#host.setModelWithProviderSessionReset(candidate); + this.#host.sessionManager.appendModelChange(candidateSelector, EPHEMERAL_MODEL_CHANGE_ROLE); + this.#host.settings.getStorage()?.recordModelUsage(candidateSelector); + this.#host.setThinkingLevel(nextThinkingLevel); + if (!this.#activeRetryFallback) { + this.#activeRetryFallback = { + role, + originalSelector: currentSelector, + originalThinkingLevel: currentThinkingLevel, + lastAppliedFallbackThinkingLevel: nextThinkingLevel, + pinned: options?.pinFallback === true, + }; + } else { + this.#activeRetryFallback.lastAppliedFallbackThinkingLevel = nextThinkingLevel; + this.#activeRetryFallback.pinned = this.#activeRetryFallback.pinned || options?.pinFallback === true; + } + await this.#host.emitSessionEvent({ + type: "retry_fallback_applied", + from: currentSelector, + to: selector.raw, + role, + }); + } + + async #tryRetryModelFallback(currentSelector: string, options?: { pinFallback?: boolean }): Promise { + const role = this.#activeRetryFallback?.role ?? this.resolveRetryFallbackRole(currentSelector); + if (!role) return false; + + for (const selector of this.findRetryFallbackCandidates(role, currentSelector)) { + if (this.isRetryFallbackSelectorSuppressed(selector)) continue; + const resolved = resolveModelOverride([selector.raw], this.#host.modelRegistry, this.#host.settings); + const candidate = resolved.model ?? this.#host.modelRegistry.find(selector.provider, selector.id); + if (!candidate) continue; + const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId()); + if (!apiKey) continue; + await this.#applyRetryFallbackCandidate(role, selector, currentSelector, options); + return true; + } + + return false; + } + + /** The active model when it is a Fireworks Fast (`-fast`) variant, else undefined. */ + #activeFireworksFastModel(): Model | undefined { + const model = this.#host.model(); + return model?.provider === "fireworks" && isFireworksFastModelId(model.id) ? model : undefined; + } + + /** + * True when the current turn failed on a Fireworks Fast (`-fast`) model in a + * way that should degrade to the reliable base (Standard) model. Fast is a + * speed-optimized router with no SLA, so any *pre-content* failure — a + * transient overload/5xx or a hard "router/model not found / unsupported" — + * is worth retrying on the base id. Skips failures the base model shares: + * context overflow (compaction's job), usage limits and auth errors (same + * account/key), and turns that already emitted a tool call (replaying would + * duplicate work). Requires the base model to exist in the registry. + */ + isFireworksFastFallbackEligible(message: AssistantMessage): boolean { + const model = this.#activeFireworksFastModel(); + if (!model) return false; + if (message.stopReason !== "error") return false; + if (message.content.some(block => block.type === "toolCall")) return false; + // A content refusal/sensitivity stop is the model's decision, not a route + // failure — switching to the base model would just re-trigger it. + if (this.isClassifierRefusal(message)) return false; + const id = this.#classifyRetryMessage(message); + if (AIError.isContextOverflow(message, model.contextWindow ?? 0)) return false; + if (AIError.is(id, AIError.Flag.UsageLimit)) return false; + if (AIError.is(id, AIError.Flag.AuthFailed)) return false; + return this.#host.modelRegistry.find("fireworks", toFireworksBaseModelId(model.id)) !== undefined; + } + + /** + * True when a turn failed with a hard (non-retryable) provider error but a + * configured `retry.fallbackChains` entry covers the active model: the same + * model is not worth retrying, yet a DIFFERENT model is a fresh chance, so + * the chain is consulted before the error becomes final. Skips failures a + * model switch cannot fix or must not replay: cancellations (abort-flavored + * errors are not model faults), context overflow (compaction's job), + * classifier refusals (chain consult is handled on the retryable path with + * `pinFallback`), and turns that already emitted a tool call (replaying + * could duplicate work). + */ + isHardErrorFallbackEligible(message: AssistantMessage): boolean { + if (message.stopReason !== "error") return false; + const model = this.#host.model(); + if (!model) return false; + const retrySettings = this.#host.settings.getGroup("retry"); + if (!retrySettings.enabled || !retrySettings.modelFallback) return false; + if (this.isClassifierRefusal(message)) return false; + const id = this.#classifyRetryMessage(message); + if (AIError.is(id, AIError.Flag.Abort) || AIError.is(id, AIError.Flag.UserInterrupt)) return false; + if (AIError.isContextOverflow(message, model.contextWindow ?? 0)) return false; + if (this.#hasReplayUnsafeToolOutput(message)) return false; + const currentSelector = formatRetryFallbackSelector(model, this.#host.thinkingLevel()); + const role = this.#activeRetryFallback?.role ?? this.resolveRetryFallbackRole(currentSelector); + if (!role) return false; + return this.findRetryFallbackCandidates(role, currentSelector).length > 0; + } + + /** + * Switch the active model from a Fireworks Fast (`-fast`) variant to its base + * (Standard) id and stick there for the rest of the session — the auto + * fallback that makes Fast a safe default. Returns false when the current + * model is not a fast variant, the base id is missing, or it has no key. + */ + async #tryFireworksFastFallback(currentSelector: string): Promise { + const model = this.#activeFireworksFastModel(); + if (!model) return false; + const baseModel = this.#host.modelRegistry.find("fireworks", toFireworksBaseModelId(model.id)); + if (!baseModel) return false; + const apiKey = await this.#host.modelRegistry.getApiKey(baseModel, this.#host.sessionId()); + if (!apiKey) return false; + const baseSelector = formatModelStringWithRouting(baseModel); + this.#host.setModelWithProviderSessionReset(baseModel); + this.#host.sessionManager.appendModelChange(baseSelector, EPHEMERAL_MODEL_CHANGE_ROLE); + this.#host.settings.getStorage()?.recordModelUsage(baseSelector); + await this.#host.emitSessionEvent({ + type: "retry_fallback_applied", + from: currentSelector, + to: baseSelector, + role: "fireworks-fast", + }); + return true; + } + + async #maybeRestoreRetryFallbackPrimary(): Promise { + if (!this.#activeRetryFallback) return; + if (this.#activeRetryFallback.pinned) return; + if (this.#getRetryFallbackRevertPolicy() !== "cooldown-expiry") return; + + const { + originalSelector: originalSelectorRaw, + originalThinkingLevel, + lastAppliedFallbackThinkingLevel, + } = this.#activeRetryFallback; + const originalSelector = parseRetryFallbackSelector(originalSelectorRaw, this.#host.modelRegistry); + if (!originalSelector) { + this.clearActiveRetryFallback(); + return; + } + + const currentModel = this.#host.model(); + if (!currentModel) return; + const currentSelector = formatRetryFallbackSelector(currentModel, this.#host.thinkingLevel()); + if (currentSelector === originalSelector.raw) { + if (!this.isRetryFallbackSelectorSuppressed(originalSelector)) { + this.clearActiveRetryFallback(); + } + return; + } + if (this.isRetryFallbackSelectorSuppressed(originalSelector)) return; + + const resolvedPrimary = resolveModelOverride( + [originalSelector.raw], + this.#host.modelRegistry, + this.#host.settings, + ); + const primaryModel = + resolvedPrimary.model ?? this.#host.modelRegistry.find(originalSelector.provider, originalSelector.id); + if (!primaryModel) return; + const apiKey = await this.#host.modelRegistry.getApiKey(primaryModel, this.#host.sessionId()); + if (!apiKey) return; + + const currentThinkingLevel = this.#host.configuredThinkingLevel(); + const thinkingToApply = + currentThinkingLevel === lastAppliedFallbackThinkingLevel ? originalThinkingLevel : currentThinkingLevel; + const primarySelector = formatModelStringWithRouting(primaryModel); + this.#host.setModelWithProviderSessionReset(primaryModel); + this.#host.sessionManager.appendModelChange(primarySelector, EPHEMERAL_MODEL_CHANGE_ROLE); + this.#host.settings.getStorage()?.recordModelUsage(primarySelector); + this.#host.setThinkingLevel(thinkingToApply); + this.clearActiveRetryFallback(); + } + + #parseRetryAfterMsFromError(errorMessage: string): number | undefined { + const now = Date.now(); + const retryAfterMsMatch = /retry-after-ms\s*[:=]\s*(\d+)/i.exec(errorMessage); + if (retryAfterMsMatch) { + return Math.max(0, Number(retryAfterMsMatch[1])); + } + + const retryAfterMatch = /retry-after\s*[:=]\s*([^\s,;]+)/i.exec(errorMessage); + if (retryAfterMatch) { + const value = retryAfterMatch[1]; + const seconds = Number(value); + if (!Number.isNaN(seconds)) { + return Math.max(0, seconds * 1000); + } + const dateMs = Date.parse(value); + if (!Number.isNaN(dateMs)) { + return Math.max(0, dateMs - now); + } + } + + const retryHintMs = extractRetryHint(undefined, errorMessage); + if (retryHintMs !== undefined) { + return retryHintMs; + } + + const resetMsMatch = /x-ratelimit-reset-ms\s*[:=]\s*(\d+)/i.exec(errorMessage); + if (resetMsMatch) { + const resetMs = Number(resetMsMatch[1]); + if (!Number.isNaN(resetMs)) { + if (resetMs > 1_000_000_000_000) { + return Math.max(0, resetMs - now); + } + return Math.max(0, resetMs); + } + } + + const resetMatch = /x-ratelimit-reset\s*[:=]\s*(\d+)/i.exec(errorMessage); + if (resetMatch) { + const resetSeconds = Number(resetMatch[1]); + if (!Number.isNaN(resetSeconds)) { + if (resetSeconds > 1_000_000_000) { + return Math.max(0, resetSeconds * 1000 - now); + } + return Math.max(0, resetSeconds * 1000); + } + } + + // Smart Fallback if no exact headers found + return undefined; + } + + /** + * Handle retryable errors with exponential backoff, credential rotation, and + * model-fallback chains. Also entered for NON-retryable errors when a switch + * is the recovery (`fireworksFastFallback`, `hardErrorFallback`): then a + * successful model switch retries immediately, and a failed switch surfaces + * the error without a same-model backoff retry. + * @returns true if retry was initiated, false if max retries exceeded or disabled + */ + async #handleRetryableError( + message: AssistantMessage, + options?: { + allowModelFallback?: boolean; + fireworksFastFallback?: boolean; + hardErrorFallback?: boolean; + preserveFailedTurn?: boolean; + }, + ): Promise { + const retrySettings = this.#host.settings.getGroup("retry"); + // The Fireworks Fast→base degrade is an intrinsic model-selection safety net, + // not a retry loop, so it runs even when the user disabled retries: it switches + // the model once and lets the base turn proceed. + if (!retrySettings.enabled && !options?.fireworksFastFallback) return false; + const classifierRefusal = this.isClassifierRefusal(message); + + const generation = this.#host.promptGeneration(); + this.#retryAttempt++; + + // Create retry promise on first attempt so waitForRetry() can await it + // Ensure only one promise exists (avoid orphaned promises from concurrent calls) + if (!this.#retryPromise) { + const { promise, resolve } = Promise.withResolvers(); + this.#retryPromise = promise; + this.#retryResolve = resolve; + } + + // All attempts on the current model are spent. Don't fail yet: the + // fallback chain below gets one last consult. Credential rotation can + // consume the entire budget without the fallback branch ever running + // (every rotation sets switchedCredential and skips it), so without + // this last resort a provider-wide usage cap never fails over to the + // configured chain. + const maxRetries = this.#isOpenRouterThinkingStreamClose(message) + ? Math.min(retrySettings.maxRetries, 1) + : retrySettings.maxRetries; + const retryBudgetExhausted = this.#retryAttempt > maxRetries; + + const errorMessage = message.errorMessage || "Unknown error"; + const id = this.#classifyRetryMessage(message); + const staleOpenAIResponsesReplayError = AIError.is(id, AIError.Flag.StaleResponsesItem); + const parsedRetryAfterMs = this.#parseRetryAfterMsFromError(errorMessage); + let delayMs = staleOpenAIResponsesReplayError + ? 0 + : calculateRetryBackoffDelayMs(retrySettings.baseDelayMs, this.#retryAttempt); + let switchedCredential = false; + let switchedModel = false; + // Set when a usage-limit error pinned the wait to credential + // availability — suppresses the generic retry-after bump below. + let usageLimitWaitMs: number | undefined; + + if (staleOpenAIResponsesReplayError) { + this.#host.resetCurrentResponsesProviderSession("stale replay error"); + } + + const activeModel = this.#host.model(); + if ( + !retryBudgetExhausted && + activeModel && + !staleOpenAIResponsesReplayError && + AIError.is(id, AIError.Flag.UsageLimit) + ) { + const retryAfterMs = parsedRetryAfterMs ?? calculateRateLimitBackoffMs(parseRateLimitReason(errorMessage)); + const outcome = await this.#host.modelRegistry.authStorage.markUsageLimitReached( + activeModel.provider, + this.#host.sessionId(), + { + retryAfterMs, + baseUrl: activeModel.baseUrl, + modelId: activeModel.id, + }, + ); + if (outcome.switched) { + switchedCredential = true; + delayMs = 0; + } else if (await this.#host.maybeAutoRedeemCodexReset()) { + // A live usage-limit 429 on the active Codex account, with a banked + // reset and the opt-in setting on: spend the reset and retry + // immediately instead of waiting out the window. Runs after the + // free sibling-switch above and before model fallback below. + switchedCredential = true; + delayMs = 0; + } else { + // No sibling credential is usable right now. Wait for whichever + // comes first: the provider's retry-after window for the current + // account, or the earliest moment a temporarily blocked sibling + // frees up (e.g. a 60s post-401 block or a 5-min usage-probe + // block) — the next attempt's getApiKey re-ranks and picks it up. + // Without this, one short-lived sibling block escalates a + // recoverable situation into the provider's multi-hour wait and + // trips the fail-fast cap below. + usageLimitWaitMs = retryAfterMs; + if (outcome.retryAtMs !== undefined) { + const siblingWaitMs = Math.max(0, outcome.retryAtMs - Date.now()) + SIBLING_UNBLOCK_BUFFER_MS; + if (siblingWaitMs < usageLimitWaitMs) { + usageLimitWaitMs = siblingWaitMs; + } + } + if (usageLimitWaitMs > delayMs) { + delayMs = usageLimitWaitMs; + } + } + } + + const allowModelFallback = options?.allowModelFallback !== false; + const currentModel = this.#host.model(); + const currentSelector = currentModel + ? formatRetryFallbackSelector(currentModel, this.#host.thinkingLevel()) + : undefined; + if (!staleOpenAIResponsesReplayError && !switchedCredential && currentSelector) { + // A refusal chain stops at the retry budget: the exhausted-attempt + // last resort is for provider failures, not classifier decisions. + if (allowModelFallback && retrySettings.modelFallback && !(retryBudgetExhausted && classifierRefusal)) { + if (!classifierRefusal) { + this.noteRetryFallbackCooldown(currentSelector, parsedRetryAfterMs, errorMessage); + } + switchedModel = await this.#tryRetryModelFallback(currentSelector, { pinFallback: classifierRefusal }); + } + // Auto fallback from a Fireworks Fast variant to its base model. Independent + // of the role-fallback setting: it's intrinsic to the Fast contract (speed + // best-effort, degrade to Standard on failure) and triggers on hard router + // errors the generic retry classifier would otherwise reject. + if (!switchedModel && allowModelFallback && options?.fireworksFastFallback) { + switchedModel = await this.#tryFireworksFastFallback(currentSelector); + } + if (switchedModel) { + delayMs = 0; + } else if (usageLimitWaitMs === undefined && parsedRetryAfterMs && parsedRetryAfterMs > delayMs) { + delayMs = parsedRetryAfterMs; + } + } + if (retryBudgetExhausted) { + if (!switchedModel) { + await this.persistTerminalEmptyErrorTurn(message); + // Max retries exceeded and no fallback model to switch to: emit + // final failure and reset. + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt: this.#retryAttempt - 1, + finalError: message.errorMessage, + }); + this.#clearPendingRecoveredRetryErrors(); + this.#retryAttempt = 0; + this.resolveRetry(); // Resolve so waitForRetry() completes + return false; + } + // The fallback model gets a fresh retry budget — leaving the spent + // counter in place would exhaust it again on its first error. + this.#retryAttempt = 1; + } + if (classifierRefusal && !switchedModel) { + // A prior attempt in this saga already announced `auto_retry_start` + // (retryAttempt was incremented for each call to this method, so > 1 + // means at least one earlier attempt started the loop) but this + // attempt is not going to retry — the saga must close with its own + // `auto_retry_end` so subscribers tracking retry-outstanding state + // (e.g. suppressing a duplicate error toast) don't stay latched on + // an announcement that never resolves. + if (this.#retryAttempt > 1) { + await this.persistTerminalEmptyErrorTurn(message); + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt: this.#retryAttempt - 1, + finalError: errorMessage, + }); + this.#clearPendingRecoveredRetryErrors(); + } + this.#retryAttempt = 0; + this.resolveRetry(); + return false; + } + // A fallback switch was the whole reason we entered (Fast→base degrade or + // a hard-error chain consult) but it could not happen (e.g. no candidate + // has a credential). Don't fall through to backing-off and retrying the + // failing model for an error the generic classifier wouldn't retry — + // surface it instead. + if ( + (options?.fireworksFastFallback || options?.hardErrorFallback) && + !switchedModel && + !this.isRetryableError(message) + ) { + // Same auto_retry_end backstop as the classifier-refusal branch above. + if (this.#retryAttempt > 1) { + await this.persistTerminalEmptyErrorTurn(message); + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt: this.#retryAttempt - 1, + finalError: errorMessage, + }); + this.#clearPendingRecoveredRetryErrors(); + } + this.#retryAttempt = 0; + this.resolveRetry(); + return false; + } + + // Fail-fast cap: if the provider asks us to wait longer than + // retry.maxDelayMs and we have no fallback credential or model to + // switch to, surface the error instead of sleeping. Defends against + // 3-hour Anthropic rate-limit windows that would otherwise leave a + // subagent (or interactive session) silently hung. The original + // assistant error message is preserved in agent state so the caller + // can act on it. + const maxDelayMs = retrySettings.maxDelayMs; + if (maxDelayMs > 0 && delayMs > maxDelayMs && !switchedCredential && !switchedModel) { + await this.persistTerminalEmptyErrorTurn(message); + const attempt = this.#retryAttempt; + this.#retryAttempt = 0; + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt, + finalError: `Provider requested ${delayMs}ms wait, exceeds retry.maxDelayMs (${maxDelayMs}ms). Original error: ${errorMessage}`, + }); + this.#clearPendingRecoveredRetryErrors(); + this.resolveRetry(); + return false; + } + + await this.#recordPendingRecoveredRetryError(message, id, { switchedCredential, switchedModel, delayMs }); + + await this.#host.emitSessionEvent({ + type: "auto_retry_start", + attempt: this.#retryAttempt, + maxAttempts: maxRetries, + delayMs, + errorMessage, + errorId: message.errorId, + }); + + // Resolved stream-stall tools have already emitted results. Keep that failed + // turn intact so continuation cannot repeat their side effects. + if (!options?.preserveFailedTurn) { + this.removeAssistantMessageFromActiveContext(message, "auto-retry"); + } + + // A thinking/response loop retried into identical context loops again. Inject a + // hidden redirect so the retried turn sees a directive to break the repeated + // pattern instead of re-sampling the same stalled reasoning. + this.#maybeInjectThinkingLoopRedirect(id); + + // Wait with exponential backoff (abortable). + const retryAbortController = new AbortController(); + this.#retryAbortController?.abort(); + this.#retryAbortController = retryAbortController; + try { + await scheduler.wait(delayMs, { signal: retryAbortController.signal }); + } catch { + if (this.#retryAbortController !== retryAbortController) { + return false; + } + // Aborted during sleep - emit end event so UI can clean up + const attempt = this.#retryAttempt; + this.#retryAttempt = 0; + this.#retryAbortController = undefined; + await this.#host.emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt, + finalError: "Retry cancelled", + }); + this.#clearPendingRecoveredRetryErrors(); + this.resolveRetry(); + return false; + } + if (this.#retryAbortController === retryAbortController) { + this.#retryAbortController = undefined; + } + + // Retry via continue() outside the agent_end event callback chain. + this.#host.scheduleAgentContinue({ delayMs: 1, generation }); + + return true; + } + + /** + * Inject a hidden redirect notice when a thinking/response loop is being retried, so + * the retried turn carries an instruction to break the repeated pattern instead of + * re-sampling the same stalled context. Injected on every {@link AIError.Flag.ThinkingLoop} + * retry (the failed assistant is dropped each attempt, so the notice does not accumulate + * unboundedly). No-op unless `id` carries the ThinkingLoop flag and the loop guard is + * enabled. The notice is generic on purpose — the detector's detail can quote raw model + * text, which must not be interpolated into a higher-priority developer message. + */ + #maybeInjectThinkingLoopRedirect(id: number): void { + if (!AIError.is(id, AIError.Flag.ThinkingLoop)) return; + if (this.#host.settings.get("model.loopGuard.enabled") !== true) return; + this.#host.agent.appendMessage({ + role: "custom", + customType: THINKING_LOOP_REDIRECT_TYPE, + content: thinkingLoopRedirectTemplate, + display: false, + attribution: "agent", + timestamp: Date.now(), + }); + this.#host.sessionManager.appendCustomMessageEntry( + THINKING_LOOP_REDIRECT_TYPE, + thinkingLoopRedirectTemplate, + false, + undefined, + "agent", + ); + } + + /** + * Cancel in-progress retry. + */ + abortRetry(): void { + this.#retryAbortController?.abort(); + // Note: _retryAttempt is reset in the catch block of _autoRetry + this.resolveRetry(); + } + + async #promptAgentWithIdleRetry(messages: AgentMessage[], options?: { toolChoice?: ToolChoice }): Promise { + const deadline = Date.now() + 30_000; + for (;;) { + try { + await this.#host.agent.prompt(messages, options); + return; + } catch (err) { + if (!(err instanceof AgentBusyError)) { + throw err; + } + if (Date.now() >= deadline) { + throw new Error("Timed out waiting for prior agent run to finish before prompting."); + } + await this.#host.agent.waitForIdle(); + } + } + } + + /** Whether auto-retry is currently in progress */ + get isRetrying(): boolean { + return this.#retryPromise !== undefined; + } + + /** Whether auto-retry is enabled */ + get autoRetryEnabled(): boolean { + return this.#host.settings.get("retry.enabled") ?? true; + } + + /** + * Toggle auto-retry setting. + */ + setAutoRetryEnabled(enabled: boolean): void { + this.#host.settings.set("retry.enabled", enabled); + } + /** + * Manually retry the last failed assistant turn. + * Removes the error message from active agent state when present and + * re-attempts with a fresh retry budget. + * + * A stream that stalls or aborts mid-tool-call ends the turn with + * `stopReason: "error" | "aborted"` and then appends one synthetic + * {@link isSyntheticToolResultMessage tool_result} per emitted tool call to + * preserve the provider's tool_use/tool_result pairing (see + * `createAbortedToolResult` in `agent-loop.ts`). Those placeholders trail the + * failed assistant turn, so the retry lookback walks back over them before + * checking the assistant message; it strips both the placeholders and the + * failed turn before re-attempting. + * + * A restored session deliberately omits failed assistant turns from provider + * context. In that case, the persisted display transcript remains the source + * of truth for whether the current branch has a retryable failed tail. + * + * @returns true if retry was initiated, false if no failed turn to retry or agent is busy + */ + async retry(): Promise { + if (this.#host.isStreaming() || this.#host.isCompacting() || this.isRetrying) return false; + + const messages = this.#host.agent.state.messages; + const activeTurnEnd = retryableAssistantTurnEnd(messages); + if (activeTurnEnd !== undefined) { + // Remove the failed/aborted assistant message plus its synthetic tool + // results (same as auto-retry does before re-attempting). + this.#host.agent.replaceMessages(messages.slice(0, activeTurnEnd - 1)); + } else { + // A restored session already dropped the failed assistant turn (and its + // paired synthetic tool results) from provider context, so the persisted + // display transcript is the source of truth for a retryable failed tail. + const transcriptMessages = this.#host.sessionManager.buildSessionContext({ transcript: true }).messages; + if (retryableAssistantTurnEnd(transcriptMessages) === undefined) return false; + } + + // Reset retry budget for a fresh attempt + this.#retryAttempt = 0; + + // Re-attempt the turn + this.#host.scheduleAgentContinue({ delayMs: 1 }); + + return true; + } +} diff --git a/packages/coding-agent/src/tools/todo.ts b/packages/coding-agent/src/tools/todo.ts index 5d0858b0c..9d387727d 100644 --- a/packages/coding-agent/src/tools/todo.ts +++ b/packages/coding-agent/src/tools/todo.ts @@ -2,7 +2,7 @@ import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallb import type { ToolExample } from "@oh-my-pi/pi-ai"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; -import { prompt } from "@oh-my-pi/pi-utils"; +import { isRecord, prompt } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import chalk from "chalk"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; @@ -34,6 +34,21 @@ export interface TodoPhase { tasks: TodoItem[]; } +/** Whether an unknown value is a persisted todo phase. */ +export function isTodoPhase(value: unknown): value is TodoPhase { + if (!isRecord(value) || typeof value.name !== "string" || !Array.isArray(value.tasks)) return false; + return value.tasks.every( + task => + isRecord(task) && + typeof task.content === "string" && + (task.status === "pending" || + task.status === "in_progress" || + task.status === "completed" || + task.status === "abandoned" || + task.status === "blocked"), + ); +} + export interface TodoCompletionTransition { phase: string; content: string; diff --git a/packages/utils/src/type-guards.ts b/packages/utils/src/type-guards.ts index 55512e02b..d43dc8bc8 100644 --- a/packages/utils/src/type-guards.ts +++ b/packages/utils/src/type-guards.ts @@ -2,6 +2,12 @@ export function isRecord(value: unknown): value is Record { return !!value && typeof value === "object" && !Array.isArray(value); } +/** Reads an own string-valued property without invoking accessors. */ +export function stringProperty(value: object, key: string): string | undefined { + const field = Object.getOwnPropertyDescriptor(value, key)?.value; + return typeof field === "string" ? field : undefined; +} + export function asRecord(value: unknown): Record | null { return isRecord(value) ? value : null; }