diff --git a/packages/coding-agent/src/advisor/delta-split.ts b/packages/coding-agent/src/advisor/delta-split.ts new file mode 100644 index 000000000..9ae122fc5 --- /dev/null +++ b/packages/coding-agent/src/advisor/delta-split.ts @@ -0,0 +1,82 @@ +// Candidate 4 (multi-message split) pure renderer, extracted for direct unit +// testing. Renders an advisor delta as MULTIPLE user messages — one per source +// message — instead of one ever-growing user message, so the provider prompt +// cache can incrementally hit each appended message. Provider caches are +// prefix-based: a single user message whose text keeps growing invalidates the +// whole message on every turn, pinning cache_read at the instructions/tools +// boundary (observed 14491 in production, 11066 in tests). Splitting into +// per-source user messages grows cache_read with the session (verified +// experimentally: 11066 → 11091 → 11112). +// +// Each source message is rendered INDEPENDENTLY via formatSessionHistoryMarkdown +// in chunked mode (shared toolResultIndex + consumedToolCallIds + watchedRoleState +// over the WHOLE delta), so toolCall/toolResult pairings resolve across chunk +// boundaries and consecutive same-role collapsing is byte-identical to the old +// single-block render. Concatenating the chunk texts reproduces the old advisor +// context exactly (equivalence-tested). +// +// The heading stays on the FIRST chunk; the WIP marker stays on the LAST chunk +// (candidate 3) so a wip/final flip never changes the stable prefix. +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai"; +import type { SecretObfuscator } from "../secrets/obfuscator"; +import { formatSessionHistoryMarkdown } from "../session/session-history-format"; + +/** Render options shared by the advisor single-block and multi-message paths. */ +export const ADVISOR_RENDER_OPTIONS = { + includeToolIntent: true, + watchedRoles: true, + expandPrimaryContext: true, + expandEditDiffs: true, +} as const; + +export interface RenderAdvisorDeltaChunksOptions { + wip: boolean; + includeThinking: boolean; + obfuscator?: SecretObfuscator; + advisorRegexSecretValues: ReadonlySet; +} + +export function renderAdvisorDeltaChunks( + delta: AgentMessage[], + opts: RenderAdvisorDeltaChunksOptions, +): AgentMessage[] | null { + if (delta.length === 0) return null; + + const resultsByCallId = new Map(); + for (const msg of delta) { + if (msg.role === "toolResult") resultsByCallId.set(msg.toolCallId, msg); + } + const consumed = new Set(); + const watchedRoleState = { lastLabel: undefined as string | undefined }; + + const renderChunk = (chunk: AgentMessage[]): string => + formatSessionHistoryMarkdown(chunk, { + ...ADVISOR_RENDER_OPTIONS, + includeThinking: opts.includeThinking, + toolResultIndex: resultsByCallId, + consumedToolCallIds: consumed, + watchedRoleState, + }); + + const heading = "### Session update"; + const chunks: AgentMessage[] = []; + for (let i = 0; i < delta.length; i++) { + let text = renderChunk([delta[i]]); + if (!text.trim()) continue; + if (opts.obfuscator) text = opts.obfuscator.obfuscate(text, opts.advisorRegexSecretValues); + if (i === 0) text = `${heading}\n\n${text}`; + chunks.push({ + role: "user", + content: [{ type: "text", text }], + timestamp: Date.now(), + } as AgentMessage); + } + if (chunks.length === 0) return null; + if (opts.wip) { + const last = chunks[chunks.length - 1]; + const blocks = (last as { content: unknown }).content as TextContent[]; + blocks[0].text += `\n\n---\n\n[in progress — more steps follow]`; + } + return chunks; +} diff --git a/packages/coding-agent/src/advisor/runtime.ts b/packages/coding-agent/src/advisor/runtime.ts index 071a6c014..85ea8373d 100644 --- a/packages/coding-agent/src/advisor/runtime.ts +++ b/packages/coding-agent/src/advisor/runtime.ts @@ -12,6 +12,7 @@ import { formatToolResultErrorPreview, PRIMARY_CONTEXT_CUSTOM_TYPES, } from "../session/session-history-format"; +import { ADVISOR_RENDER_OPTIONS, renderAdvisorDeltaChunks } from "./delta-split"; /** * Minimal slice of `Agent` the runtime drives — satisfied by pi-agent-core @@ -20,7 +21,7 @@ import { * this field after every prompt to detect a failed turn. */ export interface AdvisorAgent { - prompt(input: string): Promise; + prompt(input: string | AgentMessage[]): Promise; abort(reason?: unknown): void; reset(): void; /** @@ -231,13 +232,6 @@ const MAX_COALESCE_ROUNDS = 3; */ const MAX_QUARANTINE_RETRIES = 2; -const ADVISOR_RENDER_OPTIONS = { - includeToolIntent: true, - watchedRoles: true, - expandPrimaryContext: true, - expandEditDiffs: true, -} as const; - interface PendingDelta { text: string; rawMessages: AgentMessage[]; @@ -261,9 +255,29 @@ interface DeliveredMessage { function fingerprintMessage(message: AgentMessage): bigint | undefined { try { - const serialized = JSON.stringify(message); - if (serialized === undefined) return undefined; - return Bun.hash.wyhash(serialized); + // Field-selective fingerprint: hash every top-level field the advisor + // renderer actually reads (mirrors AppendOnlyContextManager.#messageDigest, + // issue #3406). Unrendered metadata (timestamp, usage, provider internals) + // churns on provider round-trips and would otherwise trigger a full + // transcript replay for a no-op change. Rendered fields (from + // session-history-format.ts): role, content, customType, display, isError, + // toolResult: cancelled/exitCode/output, custom: details. + const m = message as unknown as Record; + const payload = JSON.stringify({ + r: m.role ?? null, + c: m.content ?? null, + toolCallId: m.toolCallId ?? null, + toolName: m.toolName ?? null, + err: m.isError ?? null, + ct: m.customType ?? null, + disp: m.display ?? null, + cancel: m.cancelled ?? null, + exit: m.exitCode ?? null, + out: m.output ?? null, + det: m.details ?? null, + }); + if (payload === undefined) return undefined; + return Bun.hash.wyhash(payload); } catch { return undefined; } @@ -574,10 +588,123 @@ export class AdvisorRuntime { this.#includeThinking = true; } - #formatRawDelta(rawMessages: AgentMessage[], wip = false): string | null { + // Candidate 4 (multi-message split): render the Session update as MULTIPLE + // user messages — one per source message — instead of one ever-growing user + // message. Provider prompt caches are prefix-based: a single user message + // whose text keeps growing invalidates the whole message on every turn, so + // cache_read stays pinned at the instructions/tools boundary (observed + // 14491 in production, 11066 in tests). Splitting into per-source user + // messages lets the provider cache each appended message (verified + // experimentally: cache_read 11066 → 11091 → 11112 vs pinned 11066). + // + // Each source message is rendered INDEPENDENTLY via + // formatSessionHistoryMarkdown in chunked mode (shared toolResultIndex + + // consumedToolCallIds over the WHOLE delta), so a toolCall finds its + // toolResult across chunk boundaries and consecutive same-role collapsing + // is preserved. Concatenating the chunk texts with the same separator the + // old single-block render used yields byte-identical advisor context. + // Each chunk is delivered as its own user AgentMessage via a SINGLE + // Agent.prompt(AgentMessage[]) call, so the advisor model still runs ONCE + // per update (no per-message assistant turns). + #formatRawDeltaMessageChunks(preparedMessages: AgentMessage[], wip = false): AgentMessage[] | null { + // Consumes the ALREADY-prepared view from #prepareBatch: advisor custom + // messages are filtered and primary-context dedup is applied there, so + // splitting here never double-folds or leaks hidden messages. + const delta = preparedMessages; + if (delta.length === 0) return null; + + const obfuscator = this.host.obfuscator; + // Side effects the pure renderer cannot own: scrub the advisor's own + // history and refresh pending placeholder prefixes. + let discoveredNewRegexSecretValue = false; + const addRegexValues = (text: string): void => { + for (const secretValue of obfuscator?.collectRegexSecretValuesForObfuscation(text) ?? []) { + if (this.#advisorRegexSecretValues.has(secretValue)) continue; + this.#advisorRegexSecretValues.add(secretValue); + discoveredNewRegexSecretValue = true; + } + }; + const probeMd = formatSessionHistoryMarkdown(delta, { + ...ADVISOR_RENDER_OPTIONS, + includeThinking: this.#includeThinking, + }); + if (obfuscator?.hasSecrets()) { + for (const message of delta) { + if ( + message.role === "custom" && + PRIMARY_CONTEXT_CUSTOM_TYPES.has(message.customType) && + typeof message.content === "string" + ) { + addRegexValues(message.content); + } + } + addRegexValues(probeMd); + scrubAdvisorHistory(obfuscator, this.agent.state.messages, this.#advisorRegexSecretValues); + if (discoveredNewRegexSecretValue) { + this.#pending = this.#pending.map(delta => ({ + ...delta, + text: obfuscator.stripUnsafeFriendlyPlaceholderPrefixes(delta.text, this.#advisorRegexSecretValues), + })); + } + } + + // Message-level obfuscation mirrors the old #formatRawDelta path EXACTLY: + // only primary-context custom messages are mapped (tool args, details.diff, + // structured fields), because the old path's contract is whole-delta text + // obfuscation as the final pass. Expanding to every role would mint + // different placeholders and break byte-equivalence with the old render. + const renderDelta = + obfuscator?.hasSecrets() + ? delta.map(message => + message.role === "custom" && PRIMARY_CONTEXT_CUSTOM_TYPES.has(message.customType) + ? obfuscateAdvisorMessage(obfuscator, message, this.#advisorRegexSecretValues) + : message, + ) + : delta; + + const chunks = renderAdvisorDeltaChunks(renderDelta, { + wip, + includeThinking: this.#includeThinking, + obfuscator: obfuscator?.hasSecrets() ? obfuscator : undefined, + advisorRegexSecretValues: this.#advisorRegexSecretValues, + }); + return chunks; + } + + #formatRawDelta(rawMessages: AgentMessage[], wip = false, updateSeenContext = true): string | null { const delta = rawMessages .filter(message => !(message.role === "custom" && message.customType === "advisor")) - .map(message => this.#dedupContextMessage(message)); + .map(message => + updateSeenContext ? this.#dedupContextMessage(message) : this.#dedupContextMessageReadOnly(message), + ); + return this.#renderPreparedDelta(delta, wip); + } + + /** + * Preview variant of #dedupContextMessage: returns the collapse decision + * WITHOUT advancing the live #seenContext map. Used by #renderDelta so the + * preview text does not make the batch's first real delivery look like a + * re-injection. + */ + #dedupContextMessageReadOnly(msg: AgentMessage): AgentMessage { + if (msg.role !== "custom") return msg; + if (!PRIMARY_CONTEXT_CUSTOM_TYPES.has(msg.customType)) return msg; + if (typeof msg.content !== "string") return msg; + if (this.#seenContext.get(msg.customType) === msg.content) { + return { ...msg, content: "(unchanged — still in effect)" }; + } + return msg; + } + + /** + * Render already-prepared (deduped + advisor-filtered) messages to the + * single-block Session update text. Does NOT dedup again — callers that + * prepared the list must pass it here directly, and callers that prepared + * via #prepareBatch get byte-identical batch text to what the multi-message + * split consumes. + */ + #renderPreparedDelta(preparedMessages: AgentMessage[], wip = false): string | null { + const delta = preparedMessages; if (delta.length === 0) return null; const obfuscator = this.host.obfuscator; let md = formatSessionHistoryMarkdown(delta, { @@ -621,8 +748,15 @@ export class AdvisorRuntime { ); md = obfuscator.obfuscate(md, this.#advisorRegexSecretValues); } - const heading = wip ? "### Session update [in progress — more steps follow]" : "### Session update"; - return `${heading}\n\n${md}`; + // Candidate 3: keep the heading byte-identical between wip and final turns + // and put the WIP marker at the END of the batch, so a wip/final flip + // never changes the batch prefix. The provider prompt cache is + // prefix-based; a heading that flips between turns re-prefills the whole + // user message on every in-progress turn. + const heading = "### Session update"; + const mdHead = `${heading}\n\n${md}`; + if (!wip) return mdHead; + return `${mdHead}\n\n---\n\n[in progress — more steps follow]`; } #renderDelta(messages?: AgentMessage[], wip = false): Omit | null { @@ -643,12 +777,26 @@ export class AdvisorRuntime { delivered.fingerprint !== fingerprint ) { prefixChanged = true; + // Full replays are expensive (the whole transcript is re-sent and + // the provider prompt cache re-prefills from the system prompt), so + // record exactly which delivered message diverged and which + // top-level fields changed — without this the trigger is invisible. + try { + const oldMsg: Record = delivered.message as unknown as Record; + const newMsg: Record = current as unknown as Record; + const differingFields: string[] = []; + for (const key of new Set([...Object.keys(oldMsg), ...Object.keys(newMsg)])) { + if (JSON.stringify(oldMsg[key]) !== JSON.stringify(newMsg[key])) differingFields.push(key); + } + logger.debug("advisor delivered prefix changed", { index: i, role: newMsg.role, differingFields }); + } catch {} break; } delivered.message = current; } if (prefixChanged) { this.#epoch++; + logger.debug("advisor context reset", { reason: "delivered-prefix-changed", lastCount: this.#lastCount }); this.#resetAdvisorContext(true, true); } const rawMessages = all.slice(this.#lastCount); @@ -658,7 +806,11 @@ export class AdvisorRuntime { this.#deliveredPrefix.push({ message, fingerprint: fingerprintMessage(message) }); } this.#lastCount = all.length; - const text = this.#formatRawDelta(rawMessages, wip); + // Preview render: do NOT advance #seenContext — the batch's real dedup + // happens once in #prepareBatch. Advancing here would make the first + // real delivery of a re-injected primary-context message collapse to + // "(unchanged…)" (double-fold). + const text = this.#formatRawDelta(rawMessages, wip, false); return text ? { text, rawMessages, renderRevision: this.#renderRevision, wip } : null; } @@ -747,6 +899,7 @@ export class AdvisorRuntime { ): Promise<{ batch: string | null; rawMessages: AgentMessage[]; + preparedMessages: AgentMessage[]; finalTurns: number; wip: boolean; resetContext: boolean; @@ -794,10 +947,11 @@ export class AdvisorRuntime { // this already-popped raw batch so active plan/reference bodies are // restored without replaying any older primary transcript. this.#clearAdvisorContextAtCurrentCursor(); - const rerendered = this.#formatRawDelta(rawMessages, wip); + const { batch: rerendered, preparedMessages } = this.#prepareBatch(rawMessages, wip, batchText); return { batch: rerendered ?? (batchText || null), rawMessages, + preparedMessages, finalTurns: turns, wip, resetContext: true, @@ -826,11 +980,43 @@ export class AdvisorRuntime { wip = late.at(-1)!.wip; } - const batchObfuscator = this.host.obfuscator; - if (batchObfuscator?.hasSecrets()) { - batchText = batchObfuscator.stripUnsafeFriendlyPlaceholderPrefixes(batchText, this.#advisorRegexSecretValues); - } - return { batch: batchText || null, rawMessages, finalTurns: turns, wip, resetContext: false }; + // Prepare the deduped view AFTER coalescing (rawMessages is complete by + // now): filters advisor custom messages and collapses re-injected + // primary-context to "(unchanged…)". BOTH the single-block text and the + // multi-message split derive from this exact list so they never diverge. + const { batch: preparedBatch, preparedMessages } = this.#prepareBatch(rawMessages, wip, batchText); + return { + batch: preparedBatch ?? (batchText || null), + rawMessages, + preparedMessages, + finalTurns: turns, + wip, + resetContext: false, + }; + } + + /** + * Single dedup+render pass shared by every batch finalization path (normal + * and context-reset). Filters advisor custom messages, collapses re-injected + * primary-context to "(unchanged…)" via #dedupContextMessage, renders the + * single-block batch text from the SAME prepared list the multi-message + * split consumes, so the two views can never diverge. + */ + #prepareBatch( + rawMessages: AgentMessage[], + wip: boolean, + fallback: string | null, + ): { batch: string | null; preparedMessages: AgentMessage[] } { + // Dedup against the LIVE #seenContext (populated by previous turns via + // #renderDelta -> #formatRawDelta) so re-injected primary context that + // was ALREADY shown collapses to "(unchanged…)", while a FIRST delivery + // in this batch stays expanded. This pass advances the live map exactly + // once per batch — #renderDelta's text is a preview and must not set it. + const preparedMessages = rawMessages + .filter(message => !(message.role === "custom" && message.customType === "advisor")) + .map(message => this.#dedupContextMessage(message)); + const batch = this.#renderPreparedDelta(preparedMessages, wip); + return { batch: batch ?? fallback, preparedMessages }; } #terminalAssistantFailure(snapshot: number): AssistantMessage | undefined { @@ -872,8 +1058,11 @@ export class AdvisorRuntime { const epoch = this.#epoch; for (const delta of popped) { if (delta.renderRevision === this.#renderRevision) continue; - const refreshed = this.#formatRawDelta(delta.rawMessages, delta.wip); - if (refreshed) delta.text = refreshed; + // Batch text is finalized by #collectAndMaintainBatch -> #prepareBatch + // (single dedup+render pass). Refreshing here would run + // #dedupContextMessage a SECOND time and double-fold re-injected + // primary-context custom messages ("(unchanged…)" on first + // delivery). Mark the revision so the delta is not re-refreshed. delta.renderRevision = this.#renderRevision; } const recoveringOverflow = popped.some(delta => delta.overflowRecovery === true); @@ -891,7 +1080,7 @@ export class AdvisorRuntime { continue; } - const { batch, rawMessages, finalTurns, wip, resetContext } = result; + const { batch, rawMessages, preparedMessages, finalTurns, wip, resetContext } = result; if (this.disposed || batch === null) { this.#backlog = Math.max(0, this.#backlog - finalTurns); @@ -908,10 +1097,18 @@ export class AdvisorRuntime { const messageSnapshot = this.agent.state.messages.length; const contextWasFresh = resetContext || recoveringOverflow || messageSnapshot === 0; try { - // Reset the host's per-update advisor state (one-advise-per-update - // gate) and pass through whether this batch reviews partial work. this.host.beginAdvisorUpdate?.(wip); - const prompt = this.agent.prompt(batch); + // Candidate 4 (multi-message split): deliver the Session update as + // multiple user messages so the provider prompt cache can + // incrementally hit each appended message (cache_read grows with + // the session instead of staying pinned at the instructions/tools + // boundary). Falls back to the single-block string when the chunk + // renderer cannot split (e.g. empty delta). The split is + // byte-equivalent to the old single-block render (equivalence + // tested), so the advisor sees identical context. + const splitMessages = this.#formatRawDeltaMessageChunks(preparedMessages, wip); + const promptInput: string | AgentMessage[] = splitMessages ?? batch; + const prompt = this.agent.prompt(promptInput); this.#promptInFlight = prompt; try { await prompt; diff --git a/packages/coding-agent/src/session/session-advisors.ts b/packages/coding-agent/src/session/session-advisors.ts index f72615bdc..b679bbbc1 100644 --- a/packages/coding-agent/src/session/session-advisors.ts +++ b/packages/coding-agent/src/session/session-advisors.ts @@ -832,8 +832,14 @@ export class SessionAdvisors { let quarantined: string | undefined; try { quarantinedAdvisorOutput = undefined; - currentAdvisorInput = input; - await advisorAgent.prompt(input); + // Multi-message input (candidate 4) must serialize deterministically + // for quarantine source text; reuse the session history formatter + // rather than ad-hoc joins so all message kinds (text/tool/ + // custom/structured) are preserved exactly as rendered. + currentAdvisorInput = Array.isArray(input) + ? formatSessionHistoryMarkdown(input, { watchedRoles: true }) + : input; + await (Array.isArray(input) ? advisorAgent.prompt(input) : advisorAgent.prompt(input)); quarantined = quarantinedAdvisorOutput; } finally { quarantinedAdvisorOutput = undefined; diff --git a/packages/coding-agent/src/session/session-history-format.ts b/packages/coding-agent/src/session/session-history-format.ts index fda10f108..e1c8b4ac7 100644 --- a/packages/coding-agent/src/session/session-history-format.ts +++ b/packages/coding-agent/src/session/session-history-format.ts @@ -55,6 +55,14 @@ export interface HistoryFormatOptions { */ toolResultIndex?: ReadonlyMap; consumedToolCallIds?: Set; + /** + * Chunked rendering state: a mutable holder for the watched-role label + * (`**user**:` / `**agent**:`) that ended the previous chunk. Lets a caller + * formatting one logical transcript across several calls (advisor + * multi-message split) keep consecutive same-role collapsing byte-identical + * to the single-block render: pass one object across all chunk calls. + */ + watchedRoleState?: { lastLabel: string | undefined }; } /** Max length of the primary-arg summary inside `→ tool(...)` lines. */ @@ -313,7 +321,9 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History // (the watched agent emits one assistant message per tool call, so otherwise // every call repeats `**agent**:`). Cleared whenever a // non-role-labeled line is emitted so the next turn re-labels. - let lastWatchedLabel: string | undefined; + // Chunked callers seed the previous chunk's trailing label so collapsing + // stays byte-identical to the single-block render. + let lastWatchedLabel: string | undefined = opts?.watchedRoleState?.lastLabel; // Emit a watched-mode role label, collapsing consecutive same-role turns // under one label (matching the user/assistant paths). Used for the // user-attributed `!`/`$` execution lines so the advisor never reads them @@ -455,5 +465,9 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History } } + if (opts?.watchedRoleState) { + opts.watchedRoleState.lastLabel = lastWatchedLabel; + } + return `${lines.join("\n").trim()}\n`; } diff --git a/packages/coding-agent/test/advisor/advisor.test.ts b/packages/coding-agent/test/advisor/advisor.test.ts index 2f70d1d7f..bd169f6b0 100644 --- a/packages/coding-agent/test/advisor/advisor.test.ts +++ b/packages/coding-agent/test/advisor/advisor.test.ts @@ -43,6 +43,18 @@ async function settleUntil(predicate: () => boolean, timeoutMs = 2_000): Promise while (!predicate() && Date.now() < deadline) await Bun.sleep(2); } +function promptText(input: string | AgentMessage[]): string { + if (typeof input === "string") return input; + return input + .map(m => { + const c = (m as { content?: unknown }).content; + if (typeof c === "string") return c; + if (Array.isArray(c)) return c.map((b: unknown) => (b as { text?: string }).text ?? "").join("\n"); + return String(m); + }) + .join("\n"); +} + describe("advisor", () => { describe("advisor system prompt", () => { it("forbids concrete claims about tool arguments hidden from the advisor transcript", () => { @@ -863,7 +875,7 @@ describe("advisor", () => { }); describe("AdvisorRuntime", () => { - function makeAgent(promptInputs: string[]): AdvisorAgent { + function makeAgent(promptInputs: Array): AdvisorAgent { return { prompt: async input => { promptInputs.push(input); @@ -875,7 +887,7 @@ describe("advisor", () => { } it("coalesces multiple onTurnEnd calls while a prompt is in-flight", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstPromptPromise, resolve: finishFirstPrompt } = Promise.withResolvers(); const { promise: secondPromptDone, resolve: finishSecondPrompt } = Promise.withResolvers(); let promptCalls = 0; @@ -900,7 +912,7 @@ describe("advisor", () => { runtime.onTurnEnd(); await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("first"); + expect(promptText(promptInputs[0])).toContain("first"); messages.push({ role: "user", content: "second", timestamp: 2 } as AgentMessage); runtime.onTurnEnd(); @@ -910,7 +922,7 @@ describe("advisor", () => { finishFirstPrompt(); await secondPromptDone; expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("second"); + expect(promptText(promptInputs[1])).toContain("second"); }); it("waits for an in-flight review within the catch-up deadline", async () => { @@ -973,7 +985,7 @@ describe("advisor", () => { }); it("preserves the next user turn when an accepted empty stop is pruned", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "synthetic capture", synthetic: true, timestamp: 1 } as AgentMessage, @@ -1017,14 +1029,14 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await runtime.waitForCatchup(1000, 1); - const nextTurn = promptInputs.at(-1); + const nextTurn = promptText(promptInputs.at(-1) as string | AgentMessage[]); expect(nextTurn).toContain("real user instruction"); expect(nextTurn?.match(/real user instruction/g)).toHaveLength(1); expect(nextTurn?.indexOf("real user instruction")).toBeLessThan(nextTurn?.indexOf("checking files") ?? -1); }); it("coalesces late-arriving deltas into the batch after context maintenance", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstMaintainStarted, resolve: startFirstMaintain } = Promise.withResolvers(); const { promise: finishFirstMaintain, resolve: releaseFirstMaintain } = Promise.withResolvers(); const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); @@ -1065,8 +1077,8 @@ describe("advisor", () => { // Both deltas land in a single prompt — late arrival coalesced before agent.prompt(). expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("first"); - expect(promptInputs[0]).toContain("second"); + expect(promptText(promptInputs[0])).toContain("first"); + expect(promptText(promptInputs[0])).toContain("second"); // The loop re-checked maintenance for the expanded batch. expect(maintainCalls).toBe(2); }); @@ -1075,7 +1087,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+", mode: "replace" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstMaintainStarted, resolve: startFirstMaintain } = Promise.withResolvers(); const { promise: finishFirstMaintain, resolve: releaseFirstMaintain } = Promise.withResolvers(); const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); @@ -1117,8 +1129,8 @@ describe("advisor", () => { await promptStarted; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).not.toContain("TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).not.toContain("TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("caps maintainContext calls per drain cycle when arrivals never go stable", async () => { @@ -1126,7 +1138,7 @@ describe("advisor", () => { // each maintainContext call pushes a new turn (queue never goes stable on its // own). After exactly 3 calls the cap must stop coalescing, dispatch the // budgeted batch, and defer the final-round arrival to the next iteration. - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); let maintainCalls = 0; let runtime!: AdvisorRuntime; @@ -1173,7 +1185,7 @@ describe("advisor", () => { }); it("late-arriving delta that triggers reprime: full replay and correct turn accounting", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstMaintainStarted, resolve: startFirstMaintain } = Promise.withResolvers(); const { promise: finishFirstMaintain, resolve: releaseFirstMaintain } = Promise.withResolvers(); const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); @@ -1217,8 +1229,8 @@ describe("advisor", () => { // Full replay includes both turns. expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("turn1"); - expect(promptInputs[0]).toContain("turn2"); + expect(promptText(promptInputs[0])).toContain("turn1"); + expect(promptText(promptInputs[0])).toContain("turn2"); // Reprime resets the advisor agent. expect(resetCount).toBeGreaterThan(0); }); @@ -1289,7 +1301,7 @@ describe("advisor", () => { }); it("tags in-progress turns with [in progress] heading", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const updateStates: boolean[] = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); const agent: AdvisorAgent = { @@ -1313,12 +1325,12 @@ describe("advisor", () => { await promptStarted; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("[in progress — more steps follow]"); + expect(promptText(promptInputs[0])).toContain("[in progress — more steps follow]"); expect(updateStates).toEqual([true]); }); it("uses plain heading when willContinue is false or absent", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const updateStates: boolean[] = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); const agent: AdvisorAgent = { @@ -1342,13 +1354,13 @@ describe("advisor", () => { await promptStarted; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("### Session update\n"); - expect(promptInputs[0]).not.toContain("[in progress"); + expect(promptText(promptInputs[0])).toContain("### Session update\n"); + expect(promptText(promptInputs[0])).not.toContain("[in progress"); expect(updateStates).toEqual([false]); }); it("sends the batch when context maintenance fails", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); const agent: AdvisorAgent = { prompt: async input => { @@ -1373,11 +1385,11 @@ describe("advisor", () => { await promptStarted; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("first"); + expect(promptText(promptInputs[0])).toContain("first"); }); it("excludes advisor custom messages from the rendered delta", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); const agent: AdvisorAgent = { prompt: async input => { @@ -1400,15 +1412,15 @@ describe("advisor", () => { runtime.onTurnEnd(); await promptStarted; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("hello"); - expect(promptInputs[0]).not.toContain("note"); + expect(promptText(promptInputs[0])).toContain("hello"); + expect(promptText(promptInputs[0])).not.toContain("note"); }); it("obfuscates session updates before prompting the advisor", async () => { const secret = "ADVISOR_SECRET_TOKEN_123"; const obfuscator = new SecretObfuscator([{ type: "plain", content: secret }]); const placeholder = obfuscator.obfuscate(secret); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [{ role: "user", content: `token ${secret}`, timestamp: 1 } as AgentMessage]; const host: AdvisorRuntimeHost = { @@ -1422,15 +1434,15 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain(placeholder); - expect(promptInputs[0]).not.toContain(secret); + expect(promptText(promptInputs[0])).toContain(placeholder); + expect(promptText(promptInputs[0])).not.toContain(secret); }); it("redacts expanded primary context before XML escaping", async () => { const secret = "ADVISOR&SECRET123"; const obfuscator = new SecretObfuscator([{ type: "plain", content: secret }]); const placeholder = obfuscator.obfuscate(secret); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { @@ -1452,16 +1464,16 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain(placeholder); - expect(promptInputs[0]).not.toContain(secret); - expect(promptInputs[0]).not.toContain("ADVISOR&SECRET<TOKEN>123"); + expect(promptText(promptInputs[0])).toContain(placeholder); + expect(promptText(promptInputs[0])).not.toContain(secret); + expect(promptText(promptInputs[0])).not.toContain("ADVISOR&SECRET<TOKEN>123"); }); it("redacts file-mention paths before formatting", async () => { const secret = "MENTION_SECRET_TOKEN_123"; const obfuscator = new SecretObfuscator([{ type: "plain", content: secret }]); const placeholder = obfuscator.obfuscate(secret); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { @@ -1481,15 +1493,15 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain(placeholder); - expect(promptInputs[0]).not.toContain(secret); + expect(promptText(promptInputs[0])).toContain(placeholder); + expect(promptText(promptInputs[0])).not.toContain(secret); }); it("redacts nested async-result job labels before formatting", async () => { const secret = "JOB_LABEL_SECRET_TOKEN_123"; const obfuscator = new SecretObfuscator([{ type: "plain", content: secret }]); const placeholder = obfuscator.obfuscate(secret); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { @@ -1513,15 +1525,15 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain(placeholder); - expect(promptInputs[0]).not.toContain(secret); + expect(promptText(promptInputs[0])).toContain(placeholder); + expect(promptText(promptInputs[0])).not.toContain(secret); }); it("surfaces edit diff details but redacts secrets inside the diff", async () => { const secret = "DIFF_SECRET_TOKEN_123"; const obfuscator = new SecretObfuscator([{ type: "plain", content: secret }]); const placeholder = obfuscator.obfuscate(secret); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const diff = `--- a/config.ts\n+++ b/config.ts\n@@ -1 +1 @@\n-const token = "old";\n+const token = "${secret}";`; const messages: AgentMessage[] = [ @@ -1551,10 +1563,10 @@ describe("advisor", () => { expect(promptInputs).toHaveLength(1); // The diff is surfaced to the advisor (expandEditDiffs) ... - expect(promptInputs[0]).toContain("+const token ="); + expect(promptText(promptInputs[0])).toContain("+const token ="); // ... but a secret living inside details.diff is obfuscated (details now walked). - expect(promptInputs[0]).toContain(placeholder); - expect(promptInputs[0]).not.toContain(secret); + expect(promptText(promptInputs[0])).toContain(placeholder); + expect(promptText(promptInputs[0])).not.toContain(secret); }); it("does not scan tool details omitted from advisor history", async () => { @@ -1562,7 +1574,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET for later", timestamp: 1 } as AgentMessage, @@ -1591,15 +1603,15 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("does not scan advisor-hidden successful tool-result bodies", async () => { const obfuscator = new SecretObfuscator([ { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET for later", timestamp: 1 } as AgentMessage, @@ -1623,15 +1635,15 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("does not scan tool-call arguments hidden by the primary-argument preview", async () => { const obfuscator = new SecretObfuscator([ { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+", mode: "replace" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET", timestamp: 1 } as AgentMessage, @@ -1650,8 +1662,8 @@ describe("advisor", () => { }); runtime.onTurnEnd(); await runtime.waitForCatchup(1000, 1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("does not scan failed tool-result text beyond its visible preview", async () => { @@ -1659,7 +1671,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+", mode: "replace" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET", timestamp: 1 } as AgentMessage, @@ -1679,8 +1691,8 @@ describe("advisor", () => { }); runtime.onTurnEnd(); await runtime.waitForCatchup(1000, 1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("does not scan advisor-hidden execution output", async () => { @@ -1688,7 +1700,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET for later", timestamp: 1 } as AgentMessage, @@ -1718,15 +1730,15 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("does not scan execution source after the advisor preview cap", async () => { const obfuscator = new SecretObfuscator([ { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const hiddenSuffix = `${"x".repeat(120)} tok_abc123`; const messages: AgentMessage[] = [ @@ -1755,8 +1767,8 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); }); it("does not scan advisor-hidden file mention content", async () => { @@ -1764,7 +1776,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const obfuscate = vi.spyOn(obfuscator, "obfuscate"); const messages: AgentMessage[] = [ @@ -1786,8 +1798,8 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); expect(obfuscate).not.toHaveBeenCalledWith("tok_abc123", expect.anything()); }); @@ -1796,7 +1808,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const obfuscate = vi.spyOn(obfuscator, "obfuscate"); const messages: AgentMessage[] = [ @@ -1821,8 +1833,8 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("$$TOKABC123_"); - expect(promptInputs[0]).not.toContain("tok_abc123"); + expect(promptText(promptInputs[0])).toContain("$$TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); expect(obfuscate).not.toHaveBeenCalledWith("tok_abc123", expect.anything()); }); @@ -1843,7 +1855,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const diff = `--- a/config.ts\n+++ b/config.ts\n@@ -1 +1 @@\n-const token = "old";\n+const token = "tok_abc123";`; const messages: AgentMessage[] = [ @@ -1873,7 +1885,7 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - const prompt = promptInputs[0]!; + const prompt = promptText(promptInputs[0]!); expect(prompt).not.toContain("OTHERSECRET"); expect(prompt).not.toContain("tok_abc123"); // The friendly prefix is itself a normalized rendering of the @@ -1894,7 +1906,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+", mode: "replace" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [{ role: "user", content: "first tok_abc123", timestamp: 1 } as AgentMessage]; const host: AdvisorRuntimeHost = { @@ -1911,9 +1923,9 @@ describe("advisor", () => { await runtime.waitForCatchup(1000, 1); expect(promptInputs).toHaveLength(2); - expect(promptInputs[0]).not.toContain("tok_abc123"); - expect(promptInputs[1]).not.toContain("OTHERSECRET"); - expect(promptInputs[1]).not.toContain("TOKABC123_"); + expect(promptText(promptInputs[0])).not.toContain("tok_abc123"); + expect(promptText(promptInputs[1])).not.toContain("OTHERSECRET"); + expect(promptText(promptInputs[1])).not.toContain("TOKABC123_"); }); it("scrubs prior advisor prompts when a later replace regex collides with their friendly prefix", async () => { @@ -1921,14 +1933,17 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+", mode: "replace" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const firstStoredPrompt = (): string => { const message = agent.state.messages[0]; - if (message?.role !== "user" || !("content" in message) || typeof message.content !== "string") { + if (message?.role !== "user" || !("content" in message)) { throw new Error("Expected the first advisor history item to be a user prompt"); } - return message.content; + const c = (message as { content?: unknown }).content; + if (typeof c === "string") return c; + if (Array.isArray(c)) return c.map((b: unknown) => (b as { text?: string }).text ?? "").join("\n"); + throw new Error("Unexpected content shape"); }; const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET", timestamp: 1 } as AgentMessage, @@ -1942,7 +1957,7 @@ describe("advisor", () => { runtime.onTurnEnd(); await runtime.waitForCatchup(1000, 1); - agent.state.messages.push({ role: "user", content: promptInputs[0]!, timestamp: 1 } as AgentMessage); + agent.state.messages.push({ role: "user", content: promptText(promptInputs[0]!), timestamp: 1 } as AgentMessage); expect(firstStoredPrompt()).toContain("TOKABC123_"); messages.push({ role: "user", content: "later tok_abc123", timestamp: 2 } as AgentMessage); @@ -1951,7 +1966,7 @@ describe("advisor", () => { expect(promptInputs).toHaveLength(2); expect(firstStoredPrompt()).not.toContain("TOKABC123_"); - expect(promptInputs[1]).not.toContain("TOKABC123_"); + expect(promptText(promptInputs[1])).not.toContain("TOKABC123_"); }); it("redacts secrets inside assistant thinking blocks, honoring the whole-delta friendly-prefix collision set", async () => { @@ -1967,7 +1982,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const diff = `--- a/config.ts\n+++ b/config.ts\n@@ -1 +1 @@\n-const token = "old";\n+const token = "tok_abc123";`; const messages: AgentMessage[] = [ @@ -1999,7 +2014,7 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - const prompt = promptInputs[0]!; + const prompt = promptText(promptInputs[0]!); expect(prompt).toContain("_thinking:_"); expect(prompt).not.toContain("OTHERSECRET"); expect(prompt).not.toContain("tok_abc123"); @@ -2016,7 +2031,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const staleThinking = obfuscator.obfuscate("OTHERSECRET"); agent.state.messages.push({ @@ -2056,7 +2071,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET for later", timestamp: 1 } as AgentMessage, @@ -2077,7 +2092,7 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - const prompt = promptInputs[0]!; + const prompt = promptText(promptInputs[0]!); expect(prompt).not.toContain("OTHERSECRET"); // Because the image bytes were skipped by the collision pre-scan, the // plain secret's friendly-name placeholder needed no collision avoidance. @@ -2089,7 +2104,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { @@ -2110,7 +2125,7 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - const prompt = promptInputs[0]!; + const prompt = promptText(promptInputs[0]!); expect(prompt).not.toContain("OTHERSECRET"); expect(prompt).not.toContain("tok_abc123"); expect(prompt).toContain("TOKABC123_"); @@ -2121,7 +2136,7 @@ describe("advisor", () => { { type: "plain", content: "OTHERSECRET", friendlyName: "TOKABC123" }, { type: "regex", content: "tok_[a-z0-9]+" }, ]); - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "remember OTHERSECRET for later", timestamp: 1 } as AgentMessage, @@ -2144,14 +2159,14 @@ describe("advisor", () => { await Promise.resolve(); expect(promptInputs).toHaveLength(1); - const prompt = promptInputs[0]!; + const prompt = promptText(promptInputs[0]!); expect(prompt).not.toContain("OTHERSECRET"); expect(prompt).not.toContain("tok_abc123"); expect(prompt).toContain("TOKABC123_"); }); it("expands plan-mode context once, then collapses an unchanged re-injection", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstPromptDone, resolve: finishFirst } = Promise.withResolvers(); const { promise: secondPromptDone, resolve: finishSecond } = Promise.withResolvers(); let promptCalls = 0; @@ -2187,8 +2202,8 @@ describe("advisor", () => { await firstPromptDone; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain(''); - expect(promptInputs[0]).toContain("except the single plan file named below"); + expect(promptText(promptInputs[0])).toContain(''); + expect(promptText(promptInputs[0])).toContain("except the single plan file named below"); // A later turn re-injects the byte-identical rule as a fresh message object. messages.push({ @@ -2207,12 +2222,12 @@ describe("advisor", () => { await secondPromptDone; expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("unchanged — still in effect"); - expect(promptInputs[1]).not.toContain("except the single plan file named below"); + expect(promptText(promptInputs[1])).toContain("unchanged — still in effect"); + expect(promptText(promptInputs[1])).not.toContain("except the single plan file named below"); }); it("renders the watched delta with a heading, watched-role labels, and no inner ## headings", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const messages: AgentMessage[] = [ { role: "user", content: "do the thing", timestamp: 1 } as AgentMessage, @@ -2251,7 +2266,7 @@ describe("advisor", () => { runtime.onTurnEnd(); await Promise.resolve(); expect(promptInputs).toHaveLength(1); - const prompt = promptInputs[0]; + const prompt = promptText(promptInputs[0]); expect(prompt).toContain("### Session update"); expect(prompt).toContain("**user**:"); expect(prompt).toContain("**agent**:"); @@ -2263,7 +2278,7 @@ describe("advisor", () => { }); it("handles compaction shrink without prompting", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); let messages: AgentMessage[] = [ { role: "user", content: "a", timestamp: 1 } as AgentMessage, @@ -2284,7 +2299,7 @@ describe("advisor", () => { }); it("reset re-primes the advisor with the full current transcript", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: secondPromptDone, resolve: finishSecond } = Promise.withResolvers(); let promptCalls = 0; const agent: AdvisorAgent = { @@ -2306,7 +2321,7 @@ describe("advisor", () => { runtime.onTurnEnd(); await Promise.resolve(); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("aaa"); + expect(promptText(promptInputs[0])).toContain("aaa"); // Simulate a compaction: transcript replaced, then reset. messages.length = 0; @@ -2317,11 +2332,11 @@ describe("advisor", () => { await secondPromptDone; // The next turn replays the full post-compaction transcript, not just new tail. expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("summary-bbb"); + expect(promptText(promptInputs[1])).toContain("summary-bbb"); }); it("clears advisor context without replaying primary history when maintenance requests recovery", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstPromptDone, resolve: finishFirst } = Promise.withResolvers(); const { promise: secondPromptDone, resolve: finishSecond } = Promise.withResolvers(); let promptCalls = 0; @@ -2354,7 +2369,7 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await firstPromptDone; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("aaa"); + expect(promptText(promptInputs[0])).toContain("aaa"); expect(resetCount).toBe(0); shouldResetContext = true; @@ -2363,13 +2378,13 @@ describe("advisor", () => { await secondPromptDone; expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("bbb"); - expect(promptInputs[1]).not.toContain("aaa"); + expect(promptText(promptInputs[1])).toContain("bbb"); + expect(promptText(promptInputs[1])).not.toContain("aaa"); expect(resetCount).toBe(1); }); it("preserves updates queued while async maintenance resets the advisor context", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let resetCount = 0; const agent: AdvisorAgent = { prompt: async input => { @@ -2405,15 +2420,15 @@ describe("advisor", () => { await runtime.waitForCatchup(1000, 1); expect(promptInputs).toHaveLength(2); - expect(promptInputs[0]).toContain("bbb"); - expect(promptInputs[0]).not.toContain("ccc"); - expect(promptInputs[1]).toContain("ccc"); - expect(promptInputs[1]).not.toContain("bbb"); + expect(promptText(promptInputs[0])).toContain("bbb"); + expect(promptText(promptInputs[0])).not.toContain("ccc"); + expect(promptText(promptInputs[1])).toContain("ccc"); + expect(promptText(promptInputs[1])).not.toContain("bbb"); expect(resetCount).toBe(1); }); it("re-expands active primary context when maintenance clears advisor history", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent = makeAgent(promptInputs); const planRule = "Plan mode is active. You MUST remain read-only except for the approved plan file at local://PLAN.md."; @@ -2437,7 +2452,7 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await runtime.waitForCatchup(1000, 1); - expect(promptInputs[0]).toContain(planRule); + expect(promptText(promptInputs[0])).toContain(planRule); shouldResetContext = true; messages.push({ role: "user", content: "bbb", timestamp: 3 } as AgentMessage); @@ -2452,15 +2467,15 @@ describe("advisor", () => { await runtime.waitForCatchup(1000, 1); expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("bbb"); - expect(promptInputs[1]).not.toContain("aaa"); - expect(promptInputs[1]).toContain(planRule); - expect(promptInputs[1]).not.toContain("unchanged — still in effect"); + expect(promptText(promptInputs[1])).toContain("bbb"); + expect(promptText(promptInputs[1])).not.toContain("aaa"); + expect(promptText(promptInputs[1])).toContain(planRule); + expect(promptText(promptInputs[1])).not.toContain("unchanged — still in effect"); }); it("recovers a provider overflow at the current cursor without replaying primary history", async () => { const overflowMessage = "context_length_exceeded: Your input exceeds the context window of this model."; - const promptInputs: string[] = []; + const promptInputs: Array = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [{ role: "user", content: "existing advisor context", timestamp: 1 } as AgentMessage], }; @@ -2501,7 +2516,7 @@ describe("advisor", () => { expect(promptInputs).toHaveLength(2); for (const input of promptInputs) { - expect(input).toContain("overflowing-current-update"); + expect(promptText(input)).toContain("overflowing-current-update"); expect(input).not.toContain("ancient-primary-one"); expect(input).not.toContain("ancient-primary-two"); } @@ -2512,15 +2527,15 @@ describe("advisor", () => { await settleUntil(() => promptInputs.length >= 3 && runtime.backlog === 0); expect(promptInputs).toHaveLength(3); - expect(promptInputs[2]).toContain("post-recovery-update"); - expect(promptInputs[2]).not.toContain("overflowing-current-update"); - expect(promptInputs[2]).not.toContain("ancient-primary-one"); - expect(promptInputs[2]).not.toContain("ancient-primary-two"); + expect(promptText(promptInputs[2])).toContain("post-recovery-update"); + expect(promptText(promptInputs[2])).not.toContain("overflowing-current-update"); + expect(promptText(promptInputs[2])).not.toContain("ancient-primary-one"); + expect(promptText(promptInputs[2])).not.toContain("ancient-primary-two"); expect(resetCount).toBe(1); }); it("classifies structured overflow metadata before rolling back the failed turn", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [{ role: "user", content: "existing advisor context", timestamp: 1 } as AgentMessage], }; @@ -2584,7 +2599,7 @@ describe("advisor", () => { expect(promptInputs).toHaveLength(2); for (const input of promptInputs) { - expect(input).toContain("structured-current-update"); + expect(promptText(input)).toContain("structured-current-update"); expect(input).not.toContain("ancient-primary"); } expect(resetCount).toBe(1); @@ -2592,7 +2607,7 @@ describe("advisor", () => { it("drops only a double-overflowing batch and continues queued and later updates", async () => { const overflowMessage = "context_length_exceeded: Your input exceeds the context window of this model."; - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const secondAttemptStarted = Promise.withResolvers(); const finishSecondAttempt = Promise.withResolvers(); @@ -2603,7 +2618,7 @@ describe("advisor", () => { const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); - if (!input.includes("first-overflow")) { + if (!promptText(input).includes("first-overflow")) { state.error = undefined; return; } @@ -2642,12 +2657,12 @@ describe("advisor", () => { expect(failingAttempts).toBe(2); expect(promptInputs).toHaveLength(3); for (const input of promptInputs.slice(0, 2)) { - expect(input).toContain("first-overflow"); - expect(input).not.toContain("ancient-history"); + expect(promptText(input)).toContain("first-overflow"); + expect(promptText(input)).not.toContain("ancient-history"); } - expect(promptInputs[2]).toContain("queued-small-update"); - expect(promptInputs[2]).not.toContain("first-overflow"); - expect(promptInputs[2]).not.toContain("ancient-history"); + expect(promptText(promptInputs[2])).toContain("queued-small-update"); + expect(promptText(promptInputs[2])).not.toContain("first-overflow"); + expect(promptText(promptInputs[2])).not.toContain("ancient-history"); expect(failures).toHaveLength(1); expect(runtime.backlog).toBe(0); @@ -2656,11 +2671,11 @@ describe("advisor", () => { await runtime.waitForCatchup(1000, 1); expect(promptInputs).toHaveLength(4); - expect(promptInputs[3]).toContain("later-small-update"); - expect(promptInputs[3]).not.toContain("first-overflow"); + expect(promptText(promptInputs[3])).toContain("later-small-update"); + expect(promptText(promptInputs[3])).not.toContain("first-overflow"); }); it("tracks backlog and blocks until caught up", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); const { promise: promptFinish, resolve: finishPrompt } = Promise.withResolvers(); const agent: AdvisorAgent = { @@ -2755,7 +2770,7 @@ describe("advisor", () => { }); it("retries failed prompts and only decrements backlog on success", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let fail = true; const agent: AdvisorAgent = { prompt: async input => { @@ -2785,7 +2800,7 @@ describe("advisor", () => { }); it("drops backlog after 3 consecutive failures to prevent permanent stall", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -2812,7 +2827,7 @@ describe("advisor", () => { }); it("notifies the host once when consecutive prompt failures make the advisor unavailable", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; let shouldFail = true; const agent: AdvisorAgent = { @@ -2876,7 +2891,7 @@ describe("advisor", () => { // model outright ("not supported ... (code=invalid_request_error)") // failed 351 turns/hour in a shared daemon, rebuilding heavy context // every cycle. One drop cycle must latch the runtime off. - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const agent: AdvisorAgent = { prompt: async input => { @@ -2923,7 +2938,7 @@ describe("advisor", () => { }); it("halts after three transient drop cycles without an intervening success, but not across successes", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let shouldFail = true; const agent: AdvisorAgent = { prompt: async input => { @@ -3011,7 +3026,7 @@ describe("advisor", () => { // formatter bug) must neither propagate into the primary agent's // turn-end callback nor park it on the catch-up gate — and the // unrendered delta must survive for the next turn. - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -3053,8 +3068,8 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await settleUntil(() => promptInputs.length >= 2); expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("bbb-recovered"); - expect(promptInputs[1]).toContain("ccc"); + expect(promptText(promptInputs[1])).toContain("bbb-recovered"); + expect(promptText(promptInputs[1])).toContain("ccc"); runtime.dispose(); }, 10_000); @@ -3073,13 +3088,13 @@ describe("advisor", () => { ) as AgentMessage; }; - const waitForPrompts = async (prompts: string[], count: number, timeoutMs = 10_000): Promise => { + const waitForPrompts = async (prompts: Array, count: number, timeoutMs = 10_000): Promise => { const deadline = Date.now() + timeoutMs; while (prompts.length < count && Date.now() < deadline) await Bun.sleep(5); }; it("delivers a multi-MB transcript replay completely", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -3099,13 +3114,13 @@ describe("advisor", () => { await waitForPrompts(promptInputs, 1); expect(promptInputs).toHaveLength(1); // Nothing dropped: first and last transcript messages both rendered. - expect(promptInputs[0]).toContain("msg-0 "); - expect(promptInputs[0]).toContain("msg-1999 "); + expect(promptText(promptInputs[0])).toContain("msg-0 "); + expect(promptText(promptInputs[0])).toContain("msg-1999 "); runtime.dispose(); }, 20_000); it("pairs a toolCall with its non-adjacent toolResult inside one update", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -3143,16 +3158,16 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await waitForPrompts(promptInputs, 1); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("read("); + expect(promptText(promptInputs[0])).toContain("read("); // The call+result pair rendered as completed, never as a spurious // in-flight call. - expect(promptInputs[0]).toContain("⇒ ok"); - expect(promptInputs[0]).not.toContain("⇒ pending"); + expect(promptText(promptInputs[0])).toContain("⇒ ok"); + expect(promptText(promptInputs[0])).not.toContain("⇒ pending"); runtime.dispose(); }, 20_000); it("delivers a single turn carrying a multi-MB payload", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -3181,12 +3196,12 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await waitForPrompts(promptInputs, 2); expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("huge "); + expect(promptText(promptInputs[1])).toContain("huge "); runtime.dispose(); }, 20_000); it("replays the full transcript after a reset lands between renders", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -3207,13 +3222,13 @@ describe("advisor", () => { await waitForPrompts(promptInputs, 1); // The aborted pre-reset render must not have advanced the cursor: // the post-reset replay carries the whole transcript. - const replay = promptInputs.find(input => input.includes("msg-0 ") && input.includes("msg-399 ")); + const replay = promptInputs.find(input => promptText(input).includes("msg-0 ") && promptText(input).includes("msg-399 ")); expect(replay).toBeDefined(); runtime.dispose(); }, 20_000); it("delivers interleaved turns in order without loss", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -3233,8 +3248,8 @@ describe("advisor", () => { messages.push({ role: "user", content: "late-arrival tail", timestamp: 300 } as AgentMessage); runtime.onTurnEnd(messages); const deadline = Date.now() + 10_000; - while (Date.now() < deadline && !promptInputs.join("\n").includes("late-arrival tail")) await Bun.sleep(5); - const combined = promptInputs.join("\n"); + while (Date.now() < deadline && !promptInputs.map(i => promptText(i)).join("\n").includes("late-arrival tail")) await Bun.sleep(5); + const combined = promptInputs.map(i => promptText(i)).join("\n"); // Every message exactly once, ordering preserved. expect(combined).toContain("msg-0 "); expect(combined).toContain("msg-299 "); @@ -3252,7 +3267,7 @@ describe("advisor", () => { // OpenRouter ZDR `404 No endpoints available` case from #3635). The runtime // must surface that as a failed turn even though the awaited promise did // not reject. - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; let shouldFail = true; @@ -3497,7 +3512,7 @@ describe("advisor", () => { }); it("strips echoed thinking after a classifier refusal and succeeds without a notice", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; let promptCalls = 0; @@ -3557,14 +3572,14 @@ describe("advisor", () => { await settleUntil(() => runtime.backlog === 0); expect(promptInputs).toHaveLength(2); - expect(promptInputs[0]).toContain("private reasoning"); - expect(promptInputs[1]).not.toContain("private reasoning"); - expect(promptInputs[1]).toContain("answer"); + expect(promptText(promptInputs[0])).toContain("private reasoning"); + expect(promptText(promptInputs[1])).not.toContain("private reasoning"); + expect(promptText(promptInputs[1])).toContain("answer"); expect(failures).toEqual([]); }); it("surfaces a persistent classifier refusal after one stripped resend", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; const agent: AdvisorAgent = { @@ -3612,13 +3627,13 @@ describe("advisor", () => { await settleUntil(() => failures.length === 1 && runtime.backlog === 0); expect(promptInputs).toHaveLength(2); - expect(promptInputs[0]).toContain("private reasoning"); - expect(promptInputs[1]).not.toContain("private reasoning"); + expect(promptText(promptInputs[0])).toContain("private reasoning"); + expect(promptText(promptInputs[1])).not.toContain("private reasoning"); expect(failures).toHaveLength(1); }); it("degrades on a category-less refusal", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; let promptCalls = 0; @@ -3678,13 +3693,13 @@ describe("advisor", () => { await settleUntil(() => runtime.backlog === 0); expect(promptInputs).toHaveLength(2); - expect(promptInputs[0]).toContain("private reasoning"); - expect(promptInputs[1]).not.toContain("private reasoning"); + expect(promptText(promptInputs[0])).toContain("private reasoning"); + expect(promptText(promptInputs[1])).not.toContain("private reasoning"); expect(failures).toEqual([]); }); it("calls onTurnError with state.error before retrying the batch", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const turnErrors: unknown[] = []; const events: string[] = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; @@ -3726,7 +3741,7 @@ describe("advisor", () => { }); it("calls onTurnError for each consecutive failure including the dropped third turn", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const turnErrors: unknown[] = []; const failures: unknown[] = []; const events: string[] = []; @@ -3786,7 +3801,7 @@ describe("advisor", () => { }); it("continues retrying when onTurnError rejects", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const turnErrors: unknown[] = []; const events: string[] = []; const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; @@ -3830,7 +3845,7 @@ describe("advisor", () => { it("drops a terminal non-retriable assistant failure without retrying", async () => { const errorMessage = "Codex error event: Request blocked. (code=invalid_prompt)"; - const promptInputs: string[] = []; + const promptInputs: Array = []; const rollbackCalls: number[] = []; const turnErrors: unknown[] = []; const failures: unknown[] = []; @@ -3977,7 +3992,7 @@ describe("advisor", () => { it("resets advisor context after quarantining an unavailable tool response", async () => { const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; - const promptInputs: string[] = []; + const promptInputs: Array = []; const lengthsBeforePrompt: number[] = []; let resetCalls = 0; const agent: AdvisorAgent = { @@ -4038,11 +4053,11 @@ describe("advisor", () => { expect(promptInputs).toHaveLength(2); expect(lengthsBeforePrompt).toEqual([0, 0]); - expect(promptInputs[1]).toContain("aaa"); - expect(promptInputs[1]).toContain("bbb"); + expect(promptText(promptInputs[1])).toContain("aaa"); + expect(promptText(promptInputs[1])).toContain("bbb"); }); it("re-primes queued primary updates after a quarantine reset", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstPromptStarted, resolve: startFirstPrompt } = Promise.withResolvers(); const { promise: firstPrompt, reject: rejectFirstPrompt } = Promise.withResolvers(); let promptCalls = 0; @@ -4078,8 +4093,8 @@ describe("advisor", () => { await settleUntil(() => promptInputs.length >= 2 && runtime.backlog === 0); expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("aaa"); - expect(promptInputs[1]).toContain("bbb"); + expect(promptText(promptInputs[1])).toContain("aaa"); + expect(promptText(promptInputs[1])).toContain("bbb"); }); it("notifies the host after the advisor persistently quarantines its output (issue #6661)", async () => { @@ -4153,7 +4168,7 @@ describe("advisor", () => { }); it("drops the in-flight batch when a reset aborts the advisor prompt", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const { promise: firstPromptStarted, resolve: startFirstPrompt } = Promise.withResolvers(); let rejectInFlight: ((err: unknown) => void) | undefined; let promptCalls = 0; @@ -4185,7 +4200,7 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await firstPromptStarted; expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("old-conversation"); + expect(promptText(promptInputs[0])).toContain("old-conversation"); // Conversation boundary (/new): transcript replaced and the runtime reset // while the advisor prompt is still in flight. The abort that rejects the @@ -4205,12 +4220,12 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await Bun.sleep(0); expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("new-conversation"); - expect(promptInputs[1]).not.toContain("old-conversation"); + expect(promptText(promptInputs[1])).toContain("new-conversation"); + expect(promptText(promptInputs[1])).not.toContain("old-conversation"); }); it("retries the interrupted batch after a session transition rolls back", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const firstPromptStarted = Promise.withResolvers(); let rejectInFlight: ((reason?: unknown) => void) | undefined; const agent: AdvisorAgent = { @@ -4240,7 +4255,7 @@ describe("advisor", () => { runtime.resumeAfterSessionTransition(); await settleUntil(() => runtime.backlog === 0); expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("keep me"); + expect(promptText(promptInputs[1])).toContain("keep me"); }); it.each(["success", "error"] as const)( @@ -4334,7 +4349,7 @@ describe("advisor", () => { describe("AdvisorRuntime quota classification", () => { it("pauses on quota/rate-limit errors and notifies the host without retrying", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let quotaNotified = false; let failureNotified = false; const agent: AdvisorAgent = { @@ -4377,7 +4392,7 @@ describe("advisor", () => { }); it("treats 'overloaded' as a transient server error, not quota exhaustion", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const failures: unknown[] = []; const agent: AdvisorAgent = { prompt: async input => { @@ -4407,7 +4422,7 @@ describe("advisor", () => { expect(failures).toHaveLength(1); }); it("retains the failed batch in the pending queue on quota error", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let shouldFail = true; const agent: AdvisorAgent = { prompt: async input => { @@ -4434,7 +4449,7 @@ describe("advisor", () => { expect(runtime.quotaExhausted).toBe(true); expect(runtime.backlog).toBeGreaterThan(0); expect(promptInputs).toHaveLength(1); - expect(promptInputs[0]).toContain("quota-turn"); + expect(promptText(promptInputs[0])).toContain("quota-turn"); await runtime.pauseForSessionTransition(); runtime.resumeAfterSessionTransition(); @@ -4448,7 +4463,7 @@ describe("advisor", () => { runtime.onTurnEnd(messages); await Bun.sleep(0); await Bun.sleep(0); - expect(promptInputs.at(-1)).toContain("quota-turn"); + expect(promptText(promptInputs.at(-1) as string | AgentMessage[])).toContain("quota-turn"); }); it("resolves waitForCatchup immediately when quota is exhausted", async () => { @@ -4481,7 +4496,7 @@ describe("advisor", () => { expect(Date.now() - start).toBeLessThan(1000); }); it("retries once when onTurnError signals a switched sibling credential", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let firstCall = true; const agent: AdvisorAgent = { prompt: async input => { @@ -4520,7 +4535,7 @@ describe("advisor", () => { }); it("requeues when a switched retry produces no assistant response", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const state = { messages: [] as AgentMessage[] }; let callCount = 0; const agent: AdvisorAgent = { @@ -4557,7 +4572,7 @@ describe("advisor", () => { }); it("falls through to quota pause when onTurnError returns false (no sibling)", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); @@ -4589,11 +4604,11 @@ describe("advisor", () => { expect(quotaNotified).toBe(true); }); it("drops stale quota handling when reset happens during onTurnError", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); - if (input.includes("stale-turn")) { + if (promptText(input).includes("stale-turn")) { throw new Error("insufficient_quota: you have exceeded your rate limit"); } }, @@ -4635,8 +4650,8 @@ describe("advisor", () => { expect(hookInvocations).toBe(1); expect(promptInputs).toHaveLength(2); - expect(promptInputs[0]).toContain("stale-turn"); - expect(promptInputs[1]).toContain("fresh-turn"); + expect(promptText(promptInputs[0])).toContain("stale-turn"); + expect(promptText(promptInputs[1])).toContain("fresh-turn"); expect(maintenanceSignals).toHaveLength(2); expect(maintenanceSignals[0]?.aborted).toBe(true); expect(maintenanceSignals[1]?.aborted).toBe(false); @@ -4675,7 +4690,7 @@ describe("advisor", () => { expect(recoverySignal.aborted).toBe(true); }); it("uses generic failure path when switched retry hits a non-quota error", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let callCount = 0; const agent: AdvisorAgent = { prompt: async input => { @@ -4728,7 +4743,7 @@ describe("advisor", () => { }); it("marks sibling and pauses when switched retry hits a second quota error", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; let firstCall = true; const agent: AdvisorAgent = { prompt: async input => { @@ -4774,7 +4789,7 @@ describe("advisor", () => { }); it("keeps rotating while another credential is immediately available", async () => { - const promptInputs: string[] = []; + const promptInputs: Array = []; const agent: AdvisorAgent = { prompt: async input => { promptInputs.push(input); diff --git a/packages/coding-agent/test/advisor/delta-split-obfuscation.test.ts b/packages/coding-agent/test/advisor/delta-split-obfuscation.test.ts new file mode 100644 index 000000000..fa174360a --- /dev/null +++ b/packages/coding-agent/test/advisor/delta-split-obfuscation.test.ts @@ -0,0 +1,55 @@ +// Obfuscation contract for multi-message split: renderAdvisorDeltaChunks must redact +// secrets that ACTUALLY appear in rendered advisor context — toolResult +// details.diff and custom message content — matching the old single-block path. +import { describe, expect, it } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; + +import { renderAdvisorDeltaChunks } from "../../src/advisor/delta-split"; + +function chunksToText(chunks: AgentMessage[] | null): string | null { + if (!chunks) return null; + return chunks.map(c => ((c as { content: unknown }).content as { text: string }[])[0].text).join("\n"); +} + +// Fake SecretObfuscator-compatible object for the pure renderer's text pass. +function makeObfuscator() { + return { + obfuscate: (text: string) => text.replace(/SECRETVALUE123/g, "[REDACTED]"), + } as any; +} + +describe("renderAdvisorDeltaChunks obfuscation", () => { + it("redacts secrets in toolResult details.diff", () => { + const msg = { + role: "toolResult", + toolCallId: "c1", + content: "ok", + details: { diff: "--- a/x\n+++ b/x\n-SECRETVALUE123\n+new" }, + timestamp: 1, + } as unknown as AgentMessage; + const chunks = renderAdvisorDeltaChunks([msg], { + wip: false, + includeThinking: true, + obfuscator: makeObfuscator(), + advisorRegexSecretValues: new Set(), + }); + const text = chunksToText(chunks) ?? ""; + console.log("diff chunk:", JSON.stringify(text)); + expect(text).not.toContain("SECRETVALUE123"); + expect(text).toContain("[REDACTED]"); + }); + + it("redacts secrets in user message text", () => { + const msg = { role: "user", content: [{ type: "text", text: "prefix SECRETVALUE123 suffix" }], timestamp: 1 } as AgentMessage; + const chunks = renderAdvisorDeltaChunks([msg], { + wip: false, + includeThinking: true, + obfuscator: makeObfuscator(), + advisorRegexSecretValues: new Set(), + }); + const text = chunksToText(chunks) ?? ""; + console.log("user chunk:", JSON.stringify(text)); + expect(text).not.toContain("SECRETVALUE123"); + expect(text).toContain("[REDACTED]"); + }); +}); diff --git a/packages/coding-agent/test/advisor/delta-split.test.ts b/packages/coding-agent/test/advisor/delta-split.test.ts new file mode 100644 index 000000000..46f53c171 --- /dev/null +++ b/packages/coding-agent/test/advisor/delta-split.test.ts @@ -0,0 +1,96 @@ +// Direct unit tests for the multi-message-split pure renderer (src/advisor/delta-split.ts). +// Verifies: +// 1. Multi-message split is byte-equivalent to the old single-block render +// for mixed user/assistant/toolResult history. +// 2. WIP marker lands on the LAST chunk only. +// 3. Obfuscation fixture: secrets in tool-call arguments / toolResult +// details.diff are obfuscated BEFORE chunk rendering (message-level pass), +// matching the old security contract. +import { describe, expect, it } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; + +import { renderAdvisorDeltaChunks } from "../../src/advisor/delta-split"; +import { formatSessionHistoryMarkdown } from "../../src/session/session-history-format"; + +function user(text: string, ts: number): AgentMessage { + return { role: "user", content: [{ type: "text", text }], timestamp: ts } as AgentMessage; +} +function agent(text: string, ts: number): AgentMessage { + return { + role: "assistant", + content: [{ type: "text", text }], + timestamp: ts, + usage: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0, totalTokens: 2 }, + stopReason: "stop", + } as unknown as AgentMessage; +} +function toolCall(id: string, ts: number): AgentMessage { + return { + role: "assistant", + content: [{ type: "toolCall", id, name: "read", arguments: { path: "a.ts" } }], + timestamp: ts, + usage: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0, totalTokens: 2 }, + stopReason: "tool_use", + } as unknown as AgentMessage; +} +function toolResult(id: string, ts: number): AgentMessage { + return { role: "toolResult", toolCallId: id, content: "file content", timestamp: ts } as unknown as AgentMessage; +} + +const OPTS = { includeToolIntent: true, watchedRoles: true, expandPrimaryContext: true, expandEditDiffs: true, includeThinking: true } as const; + +function chunksToText(chunks: AgentMessage[] | null): string | null { + if (!chunks) return null; + return chunks.map(c => ((c as { content: unknown }).content as { text: string }[])[0].text).join("\n"); +} + +describe("renderAdvisorDeltaChunks (delta-split)", () => { + it("alternating user/agent byte-identical to single-block", () => { + const msgs = [user("first", 1), agent("a1", 2), user("second", 3), agent("a2", 4)]; + const old = "### Session update\n\n" + formatSessionHistoryMarkdown(msgs, OPTS); + const chunks = renderAdvisorDeltaChunks(msgs, { wip: false, includeThinking: true, advisorRegexSecretValues: new Set() }); + expect(chunksToText(chunks)).toBe(old); + }); + + it("consecutive same-role user byte-identical", () => { + const msgs = [user("u1", 1), user("u2", 2), agent("a", 3)]; + const old = "### Session update\n\n" + formatSessionHistoryMarkdown(msgs, OPTS); + expect(chunksToText(renderAdvisorDeltaChunks(msgs, { wip: false, includeThinking: true, advisorRegexSecretValues: new Set() }))).toBe(old); + }); + + it("toolCall + toolResult pairing byte-identical", () => { + const msgs = [toolCall("call_1", 1), toolResult("call_1", 2), user("done", 3)]; + const old = "### Session update\n\n" + formatSessionHistoryMarkdown(msgs, OPTS); + const chunks = renderAdvisorDeltaChunks(msgs, { wip: false, includeThinking: true, advisorRegexSecretValues: new Set() }); + console.log("OLD:", JSON.stringify(old)); + console.log("NEW:", JSON.stringify(chunksToText(chunks))); + expect(chunksToText(chunks)).toBe(old); + }); + + it("complex mixed history byte-identical", () => { + const msgs = [user("question", 1), agent("thinking", 2), toolCall("c2", 3), toolResult("c2", 4), agent("answer", 5), user("follow-up", 6), user("steering", 7), agent("final", 8)]; + const old = "### Session update\n\n" + formatSessionHistoryMarkdown(msgs, OPTS); + const chunks = renderAdvisorDeltaChunks(msgs, { wip: false, includeThinking: true, advisorRegexSecretValues: new Set() }); + console.log("OLD:", JSON.stringify(old)); + console.log("NEW:", JSON.stringify(chunksToText(chunks))); + expect(chunksToText(chunks)).toBe(old); + }); + + it("wip marker lands on LAST chunk only", () => { + const msgs = [user("u1", 1), agent("a1", 2), user("u2", 3)]; + const chunks = renderAdvisorDeltaChunks(msgs, { wip: true, includeThinking: true, advisorRegexSecretValues: new Set() }); + expect(chunks).not.toBeNull(); + const texts = chunks!.map(c => ((c as { content: unknown }).content as { text: string }[])[0].text); + // Marker only in the final chunk; earlier chunks unchanged. + for (let i = 0; i < texts.length - 1; i++) { + expect(texts[i]).not.toContain("[in progress"); + } + expect(texts[texts.length - 1]).toContain("[in progress — more steps follow]"); + }); + + it("splits into multiple user messages for multi-message history", () => { + const msgs = [user("u1", 1), agent("a1", 2), user("u2", 3), agent("a2", 4)]; + const chunks = renderAdvisorDeltaChunks(msgs, { wip: false, includeThinking: true, advisorRegexSecretValues: new Set() }); + expect(chunks!.length).toBeGreaterThan(1); + }); +}); diff --git a/packages/coding-agent/test/advisor/fingerprint-multi-message.test.ts b/packages/coding-agent/test/advisor/fingerprint-multi-message.test.ts new file mode 100644 index 000000000..0c3f77c91 --- /dev/null +++ b/packages/coding-agent/test/advisor/fingerprint-multi-message.test.ts @@ -0,0 +1,156 @@ +// PoC: evaluate which candidate fix prevents advisor full-transcript replays. +// Scenarios reproduce the production triggers observed in the live session +// (omp 17.2.2, omp-cop-sticky / gpt-5.6-terra): +// A. delivered message replaced by a clone differing only in unrendered +// fields (timestamp/usage) -> full-JSON fingerprint mismatch +// B. delivered message content rewritten to a `[shaken ...]` placeholder +// (auto-shake mutates in place, then rewriteEntries yields a new object) +// C. wip heading flip (## Session update [in progress ...] vs final) +// E. rendered field change (custom.display) must replay +// F. unrendered field change (usage) must NOT replay under candidate 1 +// +// formatSessionHistoryMarkdown folds consecutive user messages into one block, +// so full-vs-incremental is judged by content: `seed-body-001` is never +// mutated by scenarios A/B/F, so its presence proves the whole history was +// re-rendered (full replay); absence means only the new tail shipped. +import { describe, expect, it } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; + +import { AdvisorRuntime, type AdvisorAgent, type AdvisorRuntimeHost } from "../../src/advisor/runtime"; + +function mkMsg(role: AgentMessage["role"], text: string, timestamp: number, extra: Record = {}): AgentMessage { + return { role, content: text, timestamp, ...extra } as AgentMessage; +} + +function history(parts: string[]): AgentMessage[] { + return parts.map((text, i) => mkMsg("user", text, i + 1)); +} + +async function settle() { + for (let i = 0; i < 60; i++) await Promise.resolve(); +} + +async function runScenario( + seed: AgentMessage[], + mutate: (messages: AgentMessage[]) => void, + extraTurn: AgentMessage[], +): Promise<{ prompts: string[] }> { + const messages: AgentMessage[] = [...seed]; + const prompts: string[] = []; + const agent: AdvisorAgent = { + prompt: async (input: string) => { + prompts.push(input); + }, + abort: () => {}, + reset: () => {}, + state: { messages: [] }, + } as unknown as AdvisorAgent; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + } as unknown as AdvisorRuntimeHost; + const runtime = new AdvisorRuntime(agent, host); + runtime.onTurnEnd(); + await settle(); + mutate(messages); + messages.push(...extraTurn); + runtime.onTurnEnd(); + await settle(); + return { prompts }; +} + +function promptTextOf(input: string | AgentMessage[]): string { + if (typeof input === "string") return input; + return input + .map(m => { + const c = (m as { content?: unknown }).content; + if (typeof c === "string") return c; + if (Array.isArray(c)) return c.map((b: { text?: string }) => b.text ?? "").join("\n"); + return String(m); + }) + .join("\n"); +} + +function describeDelta(prompts: Array): { full: boolean; tailOnly: boolean } { + if (prompts.length === 0) return { full: false, tailOnly: false }; + const last = promptTextOf(prompts[prompts.length - 1]); + const full = last.includes("seed-body-001"); + const tailOnly = !full; + return { full, tailOnly }; +} + +describe("fingerprint: field-selective fingerprint (applied)", () => { + it("scenario A: timestamp-only clone replacement is INCREMENTAL (no full replay)", async () => { + const { prompts } = await runScenario( + history(["seed-body-000", "seed-body-001"]), + messages => { + messages[0] = { ...messages[0], timestamp: 999999 } as AgentMessage; + }, + [mkMsg("user", "tail-body-002", 3)], + ); + const d = describeDelta(prompts); + expect(d.full).toBe(false); + expect(d.tailOnly).toBe(true); + }); + + it("scenario B: content rewrite to shaken placeholder STILL triggers FULL replay (content is rendered)", async () => { + const { prompts } = await runScenario( + history(["seed-body-000", "seed-body-001"]), + messages => { + messages[0] = { ...messages[0], content: "[shaken ~10 tokens — recover: artifact://1 (region 1)]" } as AgentMessage; + }, + [mkMsg("user", "tail-body-002", 3)], + ); + expect(describeDelta(prompts).full).toBe(true); + }); + + it("scenario E: rendered field change (custom.display flip) triggers FULL replay", async () => { + const messages: AgentMessage[] = [ + { + role: "custom", + customType: "xdev-mount-notice", + content: "seed-body-000", + display: true, + timestamp: 1, + } as unknown as AgentMessage, + mkMsg("user", "seed-body-001", 2), + ]; + const prompts: string[] = []; + const agent: AdvisorAgent = { + prompt: async (input: string) => { + prompts.push(input); + }, + abort: () => {}, + reset: () => {}, + state: { messages: [] }, + } as unknown as AdvisorAgent; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + } as unknown as AdvisorRuntimeHost; + const runtime = new AdvisorRuntime(agent, host); + runtime.onTurnEnd(); + await settle(); + // Replace with a NEW object whose display flipped (rewriteEntries clone). + messages[0] = { ...messages[0], display: false } as unknown as AgentMessage; + messages.push(mkMsg("user", "tail-body-002", 3)); + runtime.onTurnEnd(); + await settle(); + const last = promptTextOf(prompts[prompts.length - 1]); + // display is rendered (folding gate); flipping it must re-render history. + expect(last).toContain("seed-body-001"); + }); + + it("scenario F: unrendered field change (usage) does NOT trigger replay", async () => { + const { prompts } = await runScenario( + history(["seed-body-000", "seed-body-001"]), + messages => { + messages[0] = { ...messages[0], usage: { input_tokens: 123 } } as unknown as AgentMessage; + }, + [mkMsg("user", "tail-body-002", 3)], + ); + const d = describeDelta(prompts); + expect(d.full).toBe(false); + expect(d.tailOnly).toBe(true); + }); +}); diff --git a/packages/coding-agent/test/advisor/replay-observability.test.ts b/packages/coding-agent/test/advisor/replay-observability.test.ts new file mode 100644 index 000000000..8b4ca306b --- /dev/null +++ b/packages/coding-agent/test/advisor/replay-observability.test.ts @@ -0,0 +1,67 @@ +// The advisor's full-transcript replays re-send the entire primary history and +// force the provider to re-prefill from the system prompt. Until now none of +// the reset paths logged anything, making production replay storms +// undiagnosable. These tests pin the observability contract: every reset path +// emits a structured debug event, and the delivered-prefix path reports which +// message diverged and which fields changed. +import { describe, expect, it, vi } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { logger } from "@oh-my-pi/pi-utils"; + +import { AdvisorRuntime, type AdvisorAgent, type AdvisorRuntimeHost } from "../../src/advisor/runtime"; + +function userMessage(text: string, timestamp: number): AgentMessage { + return { role: "user", content: text, timestamp } as AgentMessage; +} + +async function settle() { + for (let i = 0; i < 50; i++) await Promise.resolve(); +} + +describe("advisor context reset observability", () => { + it("logs the diverging message and differing fields when the delivered prefix changes", async () => { + const debugSpy = vi.spyOn(logger, "debug").mockImplementation(() => {}); + try { + const messages: AgentMessage[] = [userMessage("turn one body", 1), userMessage("turn two body", 2)]; + const prompts: string[] = []; + const agent: AdvisorAgent = { + prompt: async (input: string) => { + prompts.push(input); + }, + abort: () => {}, + reset: () => {}, + state: { messages: [] }, + }; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + }; + const runtime = new AdvisorRuntime(agent, host); + + runtime.onTurnEnd(); + await settle(); + + // Replace a delivered message with a changed clone, then grow the tail. + messages[0] = userMessage("turn one body EDITED", 1); + messages.push(userMessage("turn three body", 3)); + runtime.onTurnEnd(); + await settle(); + + const events = debugSpy.mock.calls.map(call => ({ message: call[0], details: call[1] })); + const divergence = events.find(event => event.message === "advisor delivered prefix changed"); + expect(divergence).toBeDefined(); + const divergenceDetails = divergence?.details as { index: number; differingFields: string[] }; + expect(divergenceDetails.index).toBe(0); + expect(divergenceDetails.differingFields).toContain("content"); + + const reset = events.find( + event => + event.message === "advisor context reset" && + (event.details as { reason: string }).reason === "delivered-prefix-changed", + ); + expect(reset).toBeDefined(); + } finally { + debugSpy.mockRestore(); + } + }); +});