From 8b92ec937e86230817d349b3454b01439646e8bb Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 10 May 2026 18:33:50 +0200 Subject: [PATCH] feat: gpt-5 harmony errata fixes - Replaced `===== ... =====` eval cell headers with `*** Begin ` / `*** End ` markers; legacy format remains renderable in HTML exports. - Replaced hashline patch grammar with `*** Begin Patch` / `*** End Patch` envelope; old inputs without the envelope are still accepted. - Extracted `sniffEvalLanguage` into a shared `sniff.ts` module reused by the parser and tool. - Added `docs/ERRATA-GPT5-HARMONY.md` and `scripts/session-stats/harmony_backtest.py` documenting and backtesting the GPT-5 Harmony-header leak defect. --- docs/ERRATA-GPT5-HARMONY.md | 205 ++++ packages/agent/CHANGELOG.md | 13 + packages/agent/src/agent-loop.ts | 121 ++- packages/agent/src/agent.ts | 9 + packages/agent/src/harmony-leak.ts | 428 ++++++++ packages/agent/src/index.ts | 1 + packages/agent/src/types.ts | 6 + .../test/fixtures/harmony-leak-corpus.json | 72 ++ packages/agent/test/harmony-leak.test.ts | 236 +++++ packages/coding-agent/CHANGELOG.md | 7 + packages/coding-agent/src/eval/eval.lark | 41 +- packages/coding-agent/src/eval/index.ts | 1 + packages/coding-agent/src/eval/parse.ts | 411 +++----- packages/coding-agent/src/eval/sniff.ts | 28 + .../src/export/html/template.generated.ts | 2 +- .../coding-agent/src/export/html/template.js | 59 +- .../coding-agent/src/hashline/constants.ts | 20 + .../coding-agent/src/hashline/grammar.lark | 43 +- packages/coding-agent/src/hashline/input.ts | 18 +- packages/coding-agent/src/hashline/parser.ts | 13 +- .../coding-agent/src/prompts/tools/eval.md | 39 +- packages/coding-agent/src/tools/eval.ts | 49 +- .../coding-agent/test/core/hashline.test.ts | 45 + .../test/core/python-prelude.test.ts | 2 +- packages/coding-agent/test/eval/parse.test.ts | 456 ++++---- scripts/session-stats/harmony_backtest.py | 991 ++++++++++++++++++ 26 files changed, 2737 insertions(+), 579 deletions(-) create mode 100644 docs/ERRATA-GPT5-HARMONY.md create mode 100644 packages/agent/src/harmony-leak.ts create mode 100644 packages/agent/test/fixtures/harmony-leak-corpus.json create mode 100644 packages/agent/test/harmony-leak.test.ts create mode 100644 packages/coding-agent/src/eval/sniff.ts create mode 100755 scripts/session-stats/harmony_backtest.py diff --git a/docs/ERRATA-GPT5-HARMONY.md b/docs/ERRATA-GPT5-HARMONY.md new file mode 100644 index 000000000..6b8cf4d27 --- /dev/null +++ b/docs/ERRATA-GPT5-HARMONY.md @@ -0,0 +1,205 @@ +# ERRATA — GPT-5 Harmony-Header Leakage + +## 1. The problem + +OpenAI frames tool calls in the Harmony chat protocol: + +``` +<|start|>assistant<|channel|>commentary to=functions.<|message|>{ARGS}<|call|> +``` + +`<|channel|>commentary to=functions.NAME` is the **routing header** — +control tokens consumed by the runtime to dispatch the call. These +tokens never appear as content under normal operation; the runtime +strips them. + +The defect: gpt-5 models occasionally emit, **as ordinary content +inside `{ARGS}`**, the **plain-text shadow** of these routing tokens — +the same characters without the `<|…|>` brackets — and continue +producing more pseudo-routing structure (channel name, body marker, +multilingual spam, fake tool-result framing). The contamination lives +inside the visible tool argument and is dispatched to the tool as if it +were intended content. + +**Critical detail.** The actual `<|start|>` / `<|channel|>` / +`<|message|>` / `<|call|>` special tokens almost never appear in tool +args. What leaks is the bracket-less spelling — `analysis to=functions.X +code …` — because OpenAI applies a logit mask suppressing the +control-token IDs inside the args region. The mass that would have gone +to those special tokens redistributes onto the un-bracketed plain-text +representation the model also learned. This makes the leak structurally +invisible to the routing parser and lands it in the tool input verbatim. + +Manifestation in tool args (real corpus example): + +``` +~ add_function(iso, ctx, ns, "installSystemChangeObserver", + os_install_system_change_observer);】【"】【analysis to=functions.edit + code above เงินไทยฟรีuser to=functions.edit code … +``` + +The leading code is real and intended. Everything after the first +non-Latin token through the next clean structural boundary is corruption. + +--- + +## 2. Observed statistics & failure modes + +Source: `~/.omp/stats.db` (`ss_tool_calls`, `ss_assistant_msgs`), through +2026-05-10. 1.05M tool calls scanned. + +### 2.1 Rate + +| Model | Leaks in tool args | Calls | per million | +|------------------|-------------------:|--------:|------------:| +| gpt-5.4 | 37 | 226,957 | 163 | +| gpt-5.3-codex | 17 | 112,243 | 151 | +| gpt-5.5 | 2 | 80,750 | 25 | +| gpt-5.2-codex | 0 | — | — | + +Plus 15 hits in assistant visible text / thinking blobs. + +### 2.2 Tool distribution + +| Tool | Hits | +|---------------------|-----:| +| `edit` | 38 | +| `eval` | 11 | +| `report_tool_issue` | 3 | +| `grep`/`read`/`search`/`yield` | 1 each | + +Concentrated in tools with free-form (non-JSON-schema) argument formats. + +### 2.3 Leak shape (deterministic) + +``` +LEAK ::= JUNK_PREFIX MARKER CHANNEL_BODY (LEAK)? +MARKER ::= "to=functions." TOOL_NAME +CHANNEL_BODY ::= " code " (SPAM | reasoning_prose | fake_tool_output)* +JUNK_PREFIX ::= (GLITCH_TOKEN | CHANNEL_WORD | NON_LATIN_RUN | "}" | "】【")+ +``` + +**Cascading is common.** Of 96 marker occurrences across 71 contaminated +records, 39 contain ≥2 markers and 7 contain ≥3 — the model emits +multiple fake `to=functions.X code …` blocks back-to-back, often with +fake `code_output\nCell N:\n…` framing between them. Once the +plain-text scaffolding is in the residual stream, the prefix now *looks +like* a fresh tool envelope start, so the macro prior over continuations +keeps voting for more scaffolding. Self-amplifying. + +### 2.4 Glitch tokens + +Single-token identifiers in `o200k_base` whose embeddings appear to be +near-init from underrepresentation in post-training. ASCII residue +immediately before the marker in the natural corpus: + +| Surface string | Single-token | Token ID | Hits in corpus | +|-------------------|:-:|---------:|---:| +| `Japgolly` | ✅ | 199,745 | 1 | +| `Jsii` | ✅ | 114,318 | (subtoken of `Jsii_commentary`) | +| `Jsii_commentary` | — (3 toks) | — | 2 | +| `changedFiles` | — (2 toks) | — | 8 | +| `RTLU` | — (2 toks) | — | 3 | + +`Japgolly` is in the last 0.13% of the vocabulary — the same family of +GitHub-corpus residue that produced `SolidGoldMagikarp` in the 2023 +GPT-2 vocabulary (Rumbelow & Watkins). `SolidGoldMagikarp` itself +tokenizes to 5 tokens in `o200k_base` — that specific token was retired, +but the class wasn't. + +For the multi-token entries, the corpus-level signature is the surface +string; the underlying glitch trigger is a sub-token (e.g. `Jsii` inside +`Jsii_commentary`). The detector list (`G` signal) keys on the surface +strings. + +Stable across unrelated sessions. Treated as a high-precision detector +signal. + +### 2.5 Channel-word leakage + +`analysis` (5), `assistant` (5), `commentary` (3), `user` (1) appear +directly preceding `to=`. Always bare words; never `<|channel|>analysis` +or any other bracketed form. Consistent with §1 — the brackets are +masked, the words are not. + +### 2.6 Non-Latin spam residue + +96 marker hits, by script: CJK 40, Cyrillic 12, Telugu/Kannada/Malayalam +18, Thai 8, Georgian 7, Armenian 7, Arabic 1. Recurring fragments are +Chinese gambling SEO (`大发时时彩`, `天天中彩票`), Georgian/Abkhaz junk, +and Thai casino spam — well-known low-quality crawl residue. + +This is the same script distribution observed in the controlled +reproduction (§7.3), independent of the prompt's natural language. + +### 2.7 Failure-mode breakdown for the `edit` tool + +The `edit` tool exists in two variants in the corpus: + +| Variant | Calls | Recovery | +|--------------------------|------:|----------| +| Patch-DSL (`@PATH`/anchor/`~payload`) | 27 | **Recoverable** by op-truncation (§3.3) | +| JSON-schema (`{path,edits:[…]}`) | 11 | **Not recoverable** — contamination is escaped *inside* JSON strings, parser accepts it cleanly, content would be written verbatim into source files | + +For Patch-DSL leaks specifically: + +- 20/27 cases: contamination on the last input line; nothing follows. +- 7/27 cases: contamination mid-input; what follows is one of: a + duplicate replay of an earlier file/anchor, intended content for a + *different* tool call (the model started its next call inline), or + pure hallucination. Post-contamination content is never trustworthy. + +### 2.8 Mechanism (confirmed) + +**Prior collapse from null-embedding glitch tokens, into a +control-token-masked basin whose mass redistributes onto the +plain-text shadow of the Harmony protocol.** + +Step by step: + +1. The model is mid-`{ARGS}` of a Harmony tool call. The runtime applies + a logit mask suppressing structural control tokens (`<|channel|>`, + `<|message|>`, `<|call|>`, `<|start|>`, `<|end|>`) inside the args + region. Without this mask, normal generation would constantly + hallucinate envelope-closes; with it, those token IDs have logit + `-∞` in args. +2. A glitch token `g` is sampled. By construction `g` was in the BPE + merge corpus but barely in LM/RL training, so its **input embedding + `e_g` ≈ near-init noise of small norm**. +3. At position t+1, the residual update `h_{t+1} ≈ LN(h_t + e_g + Attn + + MLP)` is dominated by the prefix-derived terms; the just-emitted-token + signal is effectively absent. Generation diversity normally comes + from `e_x` steering the residual into different sub-regions — + stripped here. +4. The next-token distribution therefore collapses onto the **conditional + prior over continuations of the prefix, with local conditioning + removed**. In a tool-calling rollout context, that prior is sharply + peaked on Harmony scaffolding (control tokens + routing tokens) — + that's what RL trained. +5. The mask zeros the control-token IDs. Mass redistributes onto the + **next-best continuation**: the un-bracketed surface-form spelling of + the same protocol (`analysis`, `commentary`, ` to=functions.X`, + ` code `). This spelling is unmasked because those characters are + ordinary tokens. +6. Once a few tokens of plain-text scaffolding land in the residual + stream, the prefix now resembles a fresh envelope start. The macro + prior keeps voting for more scaffolding. Cascading (§2.3) follows. +7. Multilingual spam after the marker is the same prior-collapse + continuation, drawn from the training neighborhood of the glitch + token (often ESL/auto-generated multilingual web junk — exactly the + crawl residue in §2.6). + +**Two corollaries the corpus data demanded but only the experiment +explained:** + +- **The brackets never appear** (§1, §2.5). The mask is what makes the + leak land in plain text instead of as a real envelope-close. +- **Counterintuitive grammar dependency** (§7.4). The leak is *worse* in + formats closest to OpenAI's training distribution. Off-distribution + custom grammars dampen the macro-prior basin; the official + `*** Begin Patch` format is the strongest collapse target. + +The 2023 SolidGoldMagikarp paper documented mechanism (1)+(2)+(4). The +new piece is (5): when constrained decoding masks the natural collapse +target, the mass laundered through the un-masked plain-text shadow +becomes a structurally-invisible exfiltration channel. \ No newline at end of file diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 05e728f4d..496c79caf 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -1,6 +1,19 @@ # Changelog ## [Unreleased] +### Added + +- Added `onHarmonyLeak` option on `Agent`/loop config to receive GPT-5 Harmony leak audit callbacks +- Added harmony-leak detection and audit exports to the package index for programmatic leak detection and recovery hooks + +### Changed + +- Changed OpenAI Codex model runs to detect GPT-5 Harmony protocol leakage during streaming and automatically retry or recover tool calls instead of sending contaminated arguments downstream + +### Security + +- Hardened tool-call handling against leaked `to=functions.*` protocol tails by truncating or retrying before execution +- Hardened failure handling so repeated GPT-5 Harmony leak mitigation is retried only up to two times before escalating to an explicit error ## [14.9.0] - 2026-05-10 ### Added diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index dd1611d4a..ed9a19d49 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -12,6 +12,15 @@ import { validateToolArguments, } from "@oh-my-pi/pi-ai"; import { sanitizeText } from "@oh-my-pi/pi-natives"; +import { + createHarmonyAuditEvent, + extractHarmonyRemoved, + type HarmonyDetection, + type HarmonyRecoveredToolCall, + isHarmonyLeakMitigationTarget, + recoverHarmonyToolCall, + signalListLabel, +} from "./harmony-leak"; import type { AgentContext, AgentEvent, @@ -25,6 +34,17 @@ import type { /** Sentinel returned by the abort race in `streamAssistantResponse`. */ const ABORTED: unique symbol = Symbol("agent-loop-aborted"); +class HarmonyLeakInterruption extends Error { + constructor( + readonly detection: HarmonyDetection, + readonly removed: string, + readonly recovered?: HarmonyRecoveredToolCall, + ) { + super(`Detected GPT-5 Harmony protocol leakage (${signalListLabel(detection.signals)})`); + this.name = "HarmonyLeakInterruption"; + } +} + /** * Normalize a value coming back from `tool.execute()` (or its streaming partial-update callback) * into a structurally valid {@link AgentToolResult}. @@ -255,6 +275,8 @@ async function runLoop( let firstTurn = true; // Check for steering messages at start (user may have typed while waiting) let pendingMessages: AgentMessage[] = (await config.getSteeringMessages?.()) || []; + let harmonyRetryAttempt = 0; + let harmonyTruncateResumeCount = 0; // Outer loop: continues when queued follow-up messages arrive after agent would stop while (true) { @@ -285,7 +307,44 @@ async function runLoop( } // Stream assistant response - const message = await streamAssistantResponse(currentContext, config, signal, stream, streamFn); + let recovered: HarmonyRecoveredToolCall | undefined; + let message: AssistantMessage; + try { + message = await streamAssistantResponse( + currentContext, + config, + signal, + stream, + streamFn, + harmonyRetryAttempt, + ); + harmonyRetryAttempt = 0; + harmonyTruncateResumeCount = 0; + } catch (err) { + if (!(err instanceof HarmonyLeakInterruption)) throw err; + if (err.recovered) { + if (harmonyTruncateResumeCount >= 2) { + await emitHarmonyAudit(config, err, "escalated", harmonyRetryAttempt); + throw new Error( + `GPT-5 Harmony leak recurred after truncate-and-resume recovery (${signalListLabel(err.detection.signals)}).`, + ); + } + harmonyTruncateResumeCount++; + recovered = err.recovered; + message = recovered.message; + await emitHarmonyAudit(config, err, "truncate_resume", harmonyRetryAttempt); + } else { + if (harmonyRetryAttempt >= 2) { + await emitHarmonyAudit(config, err, "escalated", harmonyRetryAttempt); + throw new Error( + `GPT-5 Harmony leak persisted after ${harmonyRetryAttempt} retries (${signalListLabel(err.detection.signals)}).`, + ); + } + await emitHarmonyAudit(config, err, "abort_retry", harmonyRetryAttempt); + harmonyRetryAttempt++; + continue; + } + } newMessages.push(message); let steeringMessagesFromExecution: AgentMessage[] | undefined; @@ -355,6 +414,23 @@ async function runLoop( stream.end(newMessages); } +async function emitHarmonyAudit( + config: AgentLoopConfig, + interruption: HarmonyLeakInterruption, + action: "truncate_resume" | "abort_retry" | "escalated", + retryN: number, +): Promise { + await config.onHarmonyLeak?.( + createHarmonyAuditEvent({ + action, + detection: interruption.detection, + model: config.model, + retryN, + removed: interruption.removed, + }), + ); +} + /** * Stream an assistant response from the LLM. * This is where AgentMessage[] gets transformed to Message[] for the LLM. @@ -365,6 +441,7 @@ async function streamAssistantResponse( signal: AbortSignal | undefined, stream: EventStream, streamFn?: StreamFn, + harmonyRetryAttempt = 0, ): Promise { // Apply context transform if configured (AgentMessage[] → AgentMessage[]) let messages = context.messages; @@ -397,33 +474,63 @@ async function streamAssistantResponse( const dynamicToolChoice = config.getToolChoice?.(); const dynamicReasoning = config.getReasoning?.(); + const harmonyMitigationEnabled = isHarmonyLeakMitigationTarget(config.model); + const harmonyAbortController = harmonyMitigationEnabled ? new AbortController() : undefined; + const requestSignal = harmonyAbortController + ? signal + ? AbortSignal.any([signal, harmonyAbortController.signal]) + : harmonyAbortController.signal + : signal; const response = await streamFunction(config.model, llmContext, { ...config, apiKey: resolvedApiKey, metadata: resolvedMetadata, toolChoice: dynamicToolChoice ?? config.toolChoice, reasoning: dynamicReasoning ?? config.reasoning, - signal, + temperature: + harmonyRetryAttempt > 0 && config.temperature !== undefined ? config.temperature + 0.05 : config.temperature, + signal: requestSignal, }); let partialMessage: AssistantMessage | null = null; let addedPartial = false; const responseIterator = response[Symbol.asyncIterator](); + + const _interruptForHarmonyLeak = (message: AssistantMessage, detection: HarmonyDetection): never => { + const recovered = recoverHarmonyToolCall(message, detection); + const removed = recovered?.removed ?? extractHarmonyRemoved(message, detection); + harmonyAbortController?.abort(); + responseIterator.return?.()?.catch(() => {}); + if (recovered) { + if (addedPartial) { + context.messages[context.messages.length - 1] = recovered.message; + } else { + context.messages.push(recovered.message); + stream.push({ type: "message_start", message: { ...recovered.message } }); + } + stream.push({ type: "message_end", message: recovered.message }); + throw new HarmonyLeakInterruption(detection, removed, recovered); + } + if (addedPartial) { + context.messages.pop(); + } + throw new HarmonyLeakInterruption(detection, removed); + }; // Set up a single abort race: register the abort listener once for the whole // stream and reuse the same race promise for every iterator.next() instead of // allocating Promise.withResolvers and add/removeEventListener per event. let abortRacePromise: Promise | undefined; let detachAbortListener: (() => void) | undefined; - if (signal) { - if (signal.aborted) { + if (requestSignal) { + if (requestSignal.aborted) { return emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); } const { promise, resolve } = Promise.withResolvers(); const onAbort = () => resolve(ABORTED); - signal.addEventListener("abort", onAbort, { once: true }); + requestSignal.addEventListener("abort", onAbort, { once: true }); abortRacePromise = promise; - detachAbortListener = () => signal.removeEventListener("abort", onAbort); + detachAbortListener = () => requestSignal.removeEventListener("abort", onAbort); } try { @@ -439,7 +546,7 @@ async function streamAssistantResponse( } else { next = await responseIterator.next(); } - if (signal?.aborted) { + if (requestSignal?.aborted) { return emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); } if (next.done) break; diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 4255cb103..450368fe5 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -21,6 +21,7 @@ import { type ToolResultMessage, } from "@oh-my-pi/pi-ai"; import { agentLoop, agentLoopContinue } from "./agent-loop"; +import type { HarmonyAuditEvent } from "./harmony-leak"; import type { AgentContext, AgentEvent, @@ -144,6 +145,11 @@ export interface AgentOptions { * Use this when abort decisions must happen before buffered events continue flowing. */ onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void; + + /** + * Called when GPT-5 Harmony protocol leakage is detected and mitigated. + */ + onHarmonyLeak?: (event: HarmonyAuditEvent) => void | Promise; /** * Custom token budgets for thinking levels (token-based providers only). */ @@ -264,6 +270,7 @@ export class Agent { #onResponse?: SimpleStreamOptions["onResponse"]; #onSseEvent?: SimpleStreamOptions["onSseEvent"]; #onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void; + #onHarmonyLeak?: (event: HarmonyAuditEvent) => void | Promise; /** Buffered Cursor tool results with text length at time of call (for correct ordering) */ #cursorToolResultBuffer: CursorToolResultEntry[] = []; @@ -304,6 +311,7 @@ export class Agent { this.#intentTracing = opts.intentTracing === true; this.#getToolChoice = opts.getToolChoice; this.#onAssistantMessageEvent = opts.onAssistantMessageEvent; + this.#onHarmonyLeak = opts.onHarmonyLeak; } /** @@ -861,6 +869,7 @@ export class Agent { transformToolCallArguments: this.#transformToolCallArguments, intentTracing: this.#intentTracing, onAssistantMessageEvent: this.#onAssistantMessageEvent, + onHarmonyLeak: this.#onHarmonyLeak, getToolChoice, getReasoning: () => this.#state.thinkingLevel, getSteeringMessages: async () => { diff --git a/packages/agent/src/harmony-leak.ts b/packages/agent/src/harmony-leak.ts new file mode 100644 index 000000000..4a85fda07 --- /dev/null +++ b/packages/agent/src/harmony-leak.ts @@ -0,0 +1,428 @@ +/** + * GPT-5 Harmony-header leakage detection and recovery. + * + * Background and policy: see `docs/ERRATA-GPT5-HARMONY.md`. This module + * implements §3 of that document: detection by signal fusion, plus a + * truncate-and-resume primitive for the `edit` tool when its input is in + * hashline DSL form. Other tools and surfaces fall through to + * abort-and-retry handled by the agent loop. + */ +import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai"; + +// Single source of truth for the marker pattern. `M` in the errata. +// Use a fresh non-global instance for `.test()` to avoid lastIndex pitfalls. +const MARKER_RE = /\bto=functions\.[A-Za-z_]\w*/g; +const HARMONY_RE = /<\|(start|end|channel|message|call|return)\|>/g; + +// Channel-word adjacency (`C`): channel/role name appearing immediately before the marker. +const CHANNEL_WORD_RE = /\b(?:analysis|commentary|assistant|user|system|developer|tool)\s+to=functions\./; + +// Glitch-token adjacency (`G`). The Japgolly literal is escaped so this regex +// source itself does not trip detection if the file is scanned (e.g. when +// editing this module via the same agent that detects). +const GLITCH_RE = /\b(?:changedFiles|RTLU|Jsii(?:_commentary)?|\x4aapgolly)\b/; + +// Body-channel cascade (`B`): marker followed by ` code` then another marker +// within 200 chars. Single regex; no manual slicing needed. +const BODY_CASCADE_RE = /to=functions\.\w+\s+code\b[\s\S]{0,200}?to=functions\./; + +// Fake-result framing (`R`): marker followed within 80 chars by Cell N: framing. +const FAKE_RESULT_RE = /to=functions\.\w+[\s\S]{0,80}?code_output\s*\nCell\s+\d+:/; + +const FENCE_RE = /^\s*(?:```+|~~~+)/; + +// Non-Latin scripts seen in the corpus: CJK + ext, Cyrillic, Thai, Georgian, +// Armenian, Kannada, Telugu, Devanagari, Arabic, Malayalam. +const SCRIPT_CLASS = + "\u3400-\u4DBF\u4E00-\u9FFF\uF900-\uFAFF\u0400-\u04FF\u0E00-\u0E7F\u10A0-\u10FF\u0530-\u058F\u0C80-\u0CFF\u0C00-\u0C7F\u0900-\u097F\u0600-\u06FF\u0D00-\u0D7F"; +const SCRIPT_RUN_RE = new RegExp(`[${SCRIPT_CLASS}]{2,}`, "u"); +const _SCRIPT_CHAR_RE = new RegExp(`[${SCRIPT_CLASS}]`, "u"); + +// Recovery registry. Each entry's parser must recognize the configured +// sentinel (per-tool, see eval/parse.ts and hashline/parser.ts) and surface +// a warning to the model so it knows to re-issue any remaining work. +// `accepts` gates on input shape: tools whose contaminated input doesn't +// match the parser's expected DSL fall through to abort-and-retry. +// +// • `edit`: hashline DSL input begins with `@`. Apply_patch envelopes +// (`*** Begin Patch …`) and JSON-schema variants are not recoverable — +// their parsers don't recognize `*** Abort`. +// • `eval`: any string is a parseable cell sequence (the parser is lenient +// and falls back to implicit-cell mode on bare strings). +interface RecoveryConfig { + sentinel: string; + accepts: (input: string) => boolean; +} +const RECOVERY_REGISTRY: Record = { + edit: { + sentinel: "\n*** Abort\n", + accepts: input => input.replace(/^\s+/, "").startsWith("@"), + }, + eval: { + sentinel: "\n*** Abort\n", + accepts: () => true, + }, +}; + +const SIGNAL_ORDER = ["M", "C", "G", "S", "B", "R", "T"] as const; + +export type HarmonySignalClass = "H" | (typeof SIGNAL_ORDER)[number]; + +export type HarmonySurface = "assistant_text" | "assistant_thinking" | "tool_arg"; + +export interface HarmonySignal { + classes: HarmonySignalClass[]; + start: number; + end: number; + text: string; +} + +export interface HarmonyDetection { + surface: HarmonySurface; + contentIndex?: number; + toolName?: string; + toolCallId?: string; + signals: HarmonySignal[]; +} + +export interface HarmonyAuditEvent { + action: "truncate_resume" | "abort_retry" | "escalated"; + surface: HarmonySurface; + signal: string; + retryN: number; + model: string; + provider: string; + toolName?: string; + removedLen: number; + removedSha8: string; + removedPreview: string; + removedBlob?: string; +} + +export interface HarmonyRecoveredToolCall { + message: AssistantMessage; + removed: string; +} + +/** + * Whether to run leak detection on responses from this model. We default-on + * for every openai-codex model rather than enumerating ids, so a future + * gpt-5.6 (or whatever) doesn't silently bypass the mitigation. Detection + * itself is cheap; the cost of missing a leak on a new model is not. + */ +export function isHarmonyLeakMitigationTarget(model: Model): boolean { + return model.provider === "openai-codex"; +} + +export function signalListLabel(signals: readonly HarmonySignal[]): string { + const seen: string[] = []; + for (const signal of signals) { + const label = signal.classes.join("+"); + if (!seen.includes(label)) seen.push(label); + } + return seen.join(",") || "none"; +} + +/** + * Detect harmony-protocol leakage in `text`. Returns undefined if clean. + * + * Trip rule: `H` alone, or `M` paired with at least one co-signal + * (`C`/`G`/`S`/`B`/`R`/`T`). Bare `M` does not trip — this document, its + * tests, and bug reports legitimately carry the marker. + * + * `parsedEnd`, when supplied, marks the byte at which a structurally valid + * tool-argument parse ends; markers strictly after it set the `T` co-signal. + * `contentIndex`/`toolName`/`toolCallId` flow through to the returned + * detection for downstream auditing. + */ +export function detectHarmonyLeak( + text: string, + surface: HarmonySurface, + options: { + parsedEnd?: number; + contentIndex?: number; + toolName?: string; + toolCallId?: string; + } = {}, +): HarmonyDetection | undefined { + const fences = computeFenceRanges(text); + const signals: HarmonySignal[] = []; + + for (const match of text.matchAll(HARMONY_RE)) { + const start = match.index ?? 0; + if (isInsideFence(fences, start)) continue; + signals.push(makeSignal(["H"], start, start + match[0].length, match[0])); + } + + for (const match of text.matchAll(MARKER_RE)) { + const start = match.index ?? 0; + if (isInsideFence(fences, start)) continue; + const end = start + match[0].length; + const classes: HarmonySignalClass[] = ["M"]; + + const adjacent = text.slice(Math.max(0, start - 64), Math.min(text.length, end + 16)); + const near = text.slice(Math.max(0, start - 16), Math.min(text.length, end + 16)); + const forward = text.slice(start, Math.min(text.length, start + 240)); + + if (CHANNEL_WORD_RE.test(adjacent)) classes.push("C"); + if (GLITCH_RE.test(near)) classes.push("G"); + if (hasScriptMismatchNear(text, start, end)) classes.push("S"); + if (BODY_CASCADE_RE.test(forward)) classes.push("B"); + if (FAKE_RESULT_RE.test(forward)) classes.push("R"); + if (options.parsedEnd !== undefined && start >= options.parsedEnd) classes.push("T"); + + // `M` alone never trips: legitimate documentation/tests carry it. + if (classes.length > 1) { + signals.push(makeSignal(classes, start, end, match[0])); + } + } + + if (signals.length === 0) return undefined; + signals.sort((a, b) => a.start - b.start || a.end - b.end); + return { + surface, + contentIndex: options.contentIndex, + toolName: options.toolName, + toolCallId: options.toolCallId, + signals, + }; +} + +/** Scan an assistant message's content blocks; return the first detection. */ +export function detectHarmonyLeakInAssistantMessage(message: AssistantMessage): HarmonyDetection | undefined { + for (let i = 0; i < message.content.length; i++) { + const block = message.content[i]; + if (block.type === "text") { + const d = detectHarmonyLeak(block.text, "assistant_text", { contentIndex: i }); + if (d) return d; + } else if (block.type === "thinking") { + const d = detectHarmonyLeak(block.thinking, "assistant_thinking", { contentIndex: i }); + if (d) return d; + } else if (block.type === "toolCall") { + const argText = getToolArgumentText(block); + if (argText !== undefined) { + const d = detectHarmonyLeak(argText, "tool_arg", { + contentIndex: i, + toolName: block.name, + toolCallId: block.id, + }); + if (d) return d; + } + } + } + return undefined; +} + +/** + * Truncate a contaminated tool call at the start of the contaminated line and + * append the tool's recovery sentinel. Returns a recovered AssistantMessage + * (containing only the cleaned tool call), a synthetic continuation user + * message asking the model to re-issue the rest, and the removed substring + * for auditing. Returns undefined when the tool is not recovery-eligible or + * the truncation would leave nothing meaningful to dispatch. + * + * `providerPayload` is dropped from the recovered message: for Codex the + * encrypted reasoning blob is opaque/signed and we cannot validate that it is + * uncontaminated. The model re-reasons on the next turn. + */ +export function recoverHarmonyToolCall( + message: AssistantMessage, + detection: HarmonyDetection, +): HarmonyRecoveredToolCall | undefined { + if (detection.surface !== "tool_arg" || detection.contentIndex === undefined) return undefined; + const block = message.content[detection.contentIndex]; + if (!block || block.type !== "toolCall") return undefined; + + const config = RECOVERY_REGISTRY[block.name]; + if (!config) return undefined; + + const input = block.arguments?.input; + if (typeof input !== "string") return undefined; + if (!config.accepts(input)) return undefined; + + const offset = detection.signals[0]?.start; + if (offset === undefined) return undefined; + + const truncated = truncateAtLineAndAppendSentinel(input, offset, config.sentinel); + if (truncated === undefined) return undefined; + + const cleanToolCall: ToolCall = { + ...block, + arguments: { ...block.arguments, input: truncated.clean }, + }; + const cleanMessage: AssistantMessage = { + ...message, + content: [cleanToolCall], + // Drop encrypted reasoning blob: opaque, possibly carries the leak forward. + providerPayload: undefined, + stopReason: "toolUse", + errorMessage: undefined, + }; + return { message: cleanMessage, removed: truncated.removed }; +} + +/** + * Return the contaminated substring from `message` for audit purposes when + * recovery is not applicable (abort path). Walks from the first detected + * signal to end-of-content within the relevant block. Returns "" if the + * detection cannot be resolved against the message. + */ +export function extractHarmonyRemoved(message: AssistantMessage, detection: HarmonyDetection): string { + if (detection.contentIndex === undefined) return ""; + const block = message.content[detection.contentIndex]; + if (!block) return ""; + const start = detection.signals[0]?.start ?? 0; + if (block.type === "text") return block.text.slice(start); + if (block.type === "thinking") return block.thinking.slice(start); + if (block.type === "toolCall") { + const text = getToolArgumentText(block); + return text ? text.slice(start) : ""; + } + return ""; +} + +export function createHarmonyAuditEvent(params: { + action: HarmonyAuditEvent["action"]; + detection: HarmonyDetection; + model: Model; + retryN: number; + removed: string; +}): HarmonyAuditEvent { + return { + action: params.action, + surface: params.detection.surface, + signal: signalListLabel(params.detection.signals), + retryN: params.retryN, + model: params.model.id, + provider: params.model.provider, + toolName: params.detection.toolName, + removedLen: params.removed.length, + removedSha8: sha8(params.removed), + removedPreview: redactedJunkPreview(params.removed), + removedBlob: Bun.env.OMP_HARMONY_DEBUG === "1" ? params.removed : undefined, + }; +} + +// ─── internals ────────────────────────────────────────────────────────────── + +function makeSignal(classes: HarmonySignalClass[], start: number, end: number, text: string): HarmonySignal { + if (classes[0] === "H") return { classes: ["H"], start, end, text }; + const sorted: HarmonySignalClass[] = []; + for (const cls of SIGNAL_ORDER) { + if (classes.includes(cls)) sorted.push(cls); + } + return { classes: sorted, start, end, text }; +} + +/** + * Precompute fenced-code-block ranges once per text. Each range is a + * [start, end) span of bytes inside any ```/~~~ fence. O(n) once instead of + * O(n) per detected match. + */ +function computeFenceRanges(text: string): Array<[number, number]> { + const ranges: Array<[number, number]> = []; + let inFence = false; + let fenceStart = 0; + let lineStart = 0; + while (lineStart <= text.length) { + const newline = text.indexOf("\n", lineStart); + const lineEnd = newline === -1 ? text.length : newline; + const line = text.slice(lineStart, lineEnd); + if (FENCE_RE.test(line)) { + if (inFence) { + ranges.push([fenceStart, lineEnd]); + inFence = false; + } else { + fenceStart = lineStart; + inFence = true; + } + } + if (newline === -1) break; + lineStart = newline + 1; + } + if (inFence) ranges.push([fenceStart, text.length]); + return ranges; +} + +function isInsideFence(ranges: Array<[number, number]>, position: number): boolean { + for (const [start, end] of ranges) { + if (position >= start && position < end) return true; + if (start > position) break; + } + return false; +} + +function hasScriptMismatchNear(text: string, start: number, end: number): boolean { + const near = text.slice(Math.max(0, start - 32), Math.min(text.length, end + 32)); + if (!SCRIPT_RUN_RE.test(near)) return false; + const surrounding = text.slice(Math.max(0, start - 200), Math.min(text.length, end + 200)); + if (surrounding.length === 0) return false; + let ascii = 0; + for (let i = 0; i < surrounding.length; i++) { + if (surrounding.charCodeAt(i) < 128) ascii++; + } + return ascii / surrounding.length >= 0.85; +} + +/** + * Tool-call argument text used for detection scanning. For tools whose args + * include a free-form `input` string we scan that directly so reported byte + * offsets line up with the original. For everything else we fall back to a + * JSON-stringified blob so detection still fires; that path's offsets are + * NOT meaningful for slicing the original args, but the recovery path gates + * on `block.arguments.input` being a string and only ever slices that. + */ +function getToolArgumentText(toolCall: ToolCall): string | undefined { + if (typeof toolCall.arguments?.input === "string") return toolCall.arguments.input; + try { + return JSON.stringify(toolCall.arguments); + } catch { + return undefined; + } +} + +function truncateAtLineAndAppendSentinel( + input: string, + offset: number, + sentinel: string, +): { clean: string; removed: string } | undefined { + const lineStart = offset <= 0 ? 0 : input.lastIndexOf("\n", offset - 1) + 1; + if (lineStart === 0) return undefined; // would cut everything + const head = input.slice(0, lineStart).replace(/\s+$/, ""); + if (head.length === 0) return undefined; + return { + clean: head + sentinel, + removed: input.slice(lineStart), + }; +} + +function sha8(text: string): string { + return new Bun.CryptoHasher("sha256").update(text).digest("hex").slice(0, 8); +} + +const PREVIEW_KEEP_RE = new RegExp(`[${SCRIPT_CLASS}\\s】【”“…」「、。]`, "u"); +const PREVIEW_TOKEN_RE = + /^(?:to=functions\.[A-Za-z_]\w*|analysis|commentary|assistant|user|system|developer|tool|changedFiles|RTLU|Jsii(?:_commentary)?|\x4aapgolly)/; + +/** + * Privacy-safe preview for the audit log: keeps marker/channel/glitch tokens, + * non-Latin script chars, and CJK punctuation; replaces everything else + * (potential source/secrets) with `·`. Sufficient to grow the glitch-token + * denylist from logs without exposing source content. Capped at 64 chars. + */ +function redactedJunkPreview(text: string): string { + const source = text.slice(0, 64); + let out = ""; + for (let i = 0; i < source.length; ) { + const tok = PREVIEW_TOKEN_RE.exec(source.slice(i)); + if (tok) { + out += tok[0]; + i += tok[0].length; + continue; + } + const ch = source[i] ?? ""; + out += PREVIEW_KEEP_RE.test(ch) ? ch : "·"; + i++; + } + return out; +} diff --git a/packages/agent/src/index.ts b/packages/agent/src/index.ts index 51b371865..919f88bee 100644 --- a/packages/agent/src/index.ts +++ b/packages/agent/src/index.ts @@ -2,6 +2,7 @@ export * from "./agent"; // Loop functions export * from "./agent-loop"; +export * from "./harmony-leak"; // Proxy utilities export * from "./proxy"; // Thinking selectors diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index b38f6ad5e..44042bdc7 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -14,6 +14,7 @@ import type { ToolResultMessage, } from "@oh-my-pi/pi-ai"; import type { Static, TSchema } from "@sinclair/typebox"; +import type { HarmonyAuditEvent } from "./harmony-leak"; /** Stream function - can return sync or Promise for async config lookup */ export type StreamFn = ( @@ -150,6 +151,11 @@ export interface AgentLoopConfig extends SimpleStreamOptions { */ onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void; + /** + * Called when GPT-5 Harmony protocol leakage is detected and mitigated. + */ + onHarmonyLeak?: (event: HarmonyAuditEvent) => void | Promise; + /** * Dynamic tool choice override, resolved per LLM call. * When set and returns a value, overrides the static `toolChoice`. diff --git a/packages/agent/test/fixtures/harmony-leak-corpus.json b/packages/agent/test/fixtures/harmony-leak-corpus.json new file mode 100644 index 000000000..8e51859d1 --- /dev/null +++ b/packages/agent/test/fixtures/harmony-leak-corpus.json @@ -0,0 +1,72 @@ +{ + "_note": "Extracted from ~/.omp/stats.db ss_tool_calls. Positives are real contaminated tool args observed in production. Negatives are crafted to exercise the false-positive guards in the trip rule.", + "positives": [ + { + "id": "rid1096312", + "kind": "edit_dsl", + "expectation": "recover", + "input": "@packages/fxe-ui/src/components/Pressable.ts\n=89hm\n~ }ենց to=functions.edit code wrong\n", + "argJson": null + }, + { + "id": "rid1094278", + "kind": "edit_dsl", + "expectation": "recover", + "input": "@src/runtime/v8/native/async_hooks.hpp\n=5pk..8xh\n~namespace fxe::runtime {\n~ void install_native_async_hooks(v8::Isolate* iso, v8::Local ctx);\n~} // namespace fxe::runtime\n\n@src/js/v8_host.cpp\n<81cb\n~namespace fxe::runtime {\n~ void uninstall_native_async_hooks(v8::Isolate* iso);\n~}\n~大发官网 to=functions.edit code ՞նչ. Let's see. \n", + "argJson": null + }, + { + "id": "rid1094797", + "kind": "edit_dsl", + "expectation": "recover", + "input": "@src/os/os.hpp\n+50md\n~ // Install a change observer for system preferences. Callback fires on the\n~ // main thread when a known preference flips. Returns true if any observers\n~ // installed; false means callers should fall back to polling.\n~ bool install_system_change_observer(std::function cb);\n@src/os/macos/os_macos.mm\n+40th\n~\n~namespace {\n~ std::mutex g_system_change_observer_mu;\n~ std::function g_system_change_cb;\n~ bool g_system_change_observer_installed = false;\n~\n~ void emit_system_change(const char* kind) {\n~ if (!kind || *kind == '\\0')\n~ return;\n~ std::function cb;\n~ {\n~ std::lock_guard lock(g_system_change_observer_mu);\n~ cb = g_system_change_cb;\n~ }\n~ if (!cb)\n~ return;\n~ if ([NSThread isMainThread]) {\n~ cb(kind);\n~ return;\n~ }\n~ dispatch_async(dispatch_get_main_queue(), ^{\n~ std::function main_cb;\n~ {\n~ std::lock_guard lock(g_system_change_observer_mu);\n~ main_cb = g_system_change_cb;\n~ }\n~ if (main_cb)\n~ main_cb(kind);\n~ });\n~ }\n~}\n~\n~@interface FxeSystemObserver : NSObject\n~+ (instancetype)shared;\n~- (void)appearanceChanged:(NSNotification*)notification;\n~- (void)accessibilityDisplayOptionsChanged:(NSNotification*)notification;\n~- (void)systemColorsChanged:(NSNotification*)notification;\n~@end\n~\n~@implementation FxeSystemObserver\n~+ (instancetype)shared {\n~ static FxeSystemObserver* observer = [[FxeSystemObserver alloc] init];\n~ return observer;\n~}\n~\n~- (void)appearanceChanged:(NSNotification*)notification {\n~ (void)notification;\n~ emit_system_change(\"colorScheme\");\n~}\n~\n~- (void)accessibilityDisplayOptionsChanged:(NSNotification*)notification {\n~ (void)notification;\n~ emit_system_change(\"prefersReducedMotion\");\n~ emit_system_change(\"prefersHighContrast\");\n~}\n~\n~- (void)systemColorsChanged:(NSNotification*)notification {\n~ (void)notification;\n~ emit_system_change(\"accentColor\");\n~}\n~@end\n~\n+258th\n~\n~ bool install_system_change_observer(std::function cb) {\n~ {\n~ std::lock_guard lock(g_system_change_observer_mu);\n~ g_system_change_cb = std::move(cb);\n~ }\n~ __block bool installed = false;\n~ run_on_main_sync_void(^{\n~ @autoreleasepool {\n~ @try {\n~ if (!g_system_change_observer_installed) {\n~ FxeSystemObserver* observer = [FxeSystemObserver shared];\n~ [[NSDistributedNotificationCenter defaultCenter]\n~ addObserver:observer\n~ selector:@selector(appearanceChanged:)\n~ name:@\"AppleInterfaceThemeChangedNotification\"\n~ object:nil];\n~ [[[NSWorkspace sharedWorkspace] notificationCenter]\n~ addObserver:observer\n~ selector:@selector(accessibilityDisplayOptionsChanged:)\n~ name:NSWorkspaceAccessibilityDisplayOptionsDidChangeNotification\n~ object:nil];\n~ [[NSNotificationCenter defaultCenter]\n~ addObserver:observer\n~ selector:@selector(systemColorsChanged:)\n~ name:NSSystemColorsDidChangeNotification\n~ object:nil];\n~ g_system_change_observer_installed = true;\n~ }\n~ installed = true;\n~ } @catch (NSException*) {\n~ installed = false;\n~ }\n~ }\n~ });\n~ return installed;\n~ }\n~\n@src/os/linux/os_linux.cpp\n+2141st\n~\n~ bool install_system_change_observer(std::function cb) {\n~ (void)cb;\n~ return false;\n~ }\n~\n@src/os/win32/os_win32.cpp\n+1506th\n~\n~ bool install_system_change_observer(std::function cb) {\n~ (void)cb;\n~ return false;\n~ }\n~\n@src/runtime/v8/fxe_native.cpp\n+4ip\n~#include \"os/os.hpp\"\n+2551st\n~\n~ struct system_change_listener {\n~ Isolate* isolate = nullptr;\n~ Global context;\n~ Global callback;\n~ };\n~\n~ std::mutex g_system_change_listener_mu;\n~ std::unordered_map g_system_change_listeners;\n~ std::atomic g_next_system_change_listener_id{1};\n~\n~ void dispatch_system_change_listener(u32 id, std::string_view kind) {\n~ Isolate* iso = nullptr;\n~ {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it == g_system_change_listeners.end() || it->second.callback.IsEmpty() ||\n~ it->second.context.IsEmpty()) {\n~ return;\n~ }\n~ iso = it->second.isolate;\n~ }\n~ if (!iso)\n~ return;\n~ Isolate::Scope isolate_scope(iso);\n~ HandleScope handle_scope(iso);\n~ Local ctx;\n~ Local cb;\n~ {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it == g_system_change_listeners.end() || it->second.callback.IsEmpty() ||\n~ it->second.context.IsEmpty()) {\n~ return;\n~ }\n~ ctx = it->second.context.Get(iso);\n~ cb = it->second.callback.Get(iso);\n~ }\n~ if (ctx.IsEmpty() || cb.IsEmpty())\n~ return;\n~ Context::Scope context_scope(ctx);\n~ Local argv[1] = {str(iso, kind)};\n~ TryCatch try_catch(iso);\n~ Local ignored;\n~ (void)cb->Call(ctx, ctx->Global(), 1, argv).ToLocal(&ignored);\n~ }\n~\n~ void dispose_system_change_observer(const FunctionCallbackInfo& info) {\n~ auto* iso = info.GetIsolate();\n~ const u32 id = info.Data().IsEmpty()\n~ ? 0\n~ : info.Data()->Uint32Value(iso->GetCurrentContext()).FromMaybe(0);\n~ if (id == 0)\n~ return;\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it == g_system_change_listeners.end())\n~ return;\n~ it->second.callback.Reset();\n~ it->second.context.Reset();\n~ g_system_change_listeners.erase(it);\n~ }\n~\n~ void os_install_system_change_observer(const FunctionCallbackInfo& info) {\n~ auto* iso = info.GetIsolate();\n~ auto ctx = iso->GetCurrentContext();\n~ if (info.Length() < 1 || !info[0]->IsFunction()) {\n~ throw_error(iso, \"__fxe_native.os.installSystemChangeObserver requires a callback\");\n~ return;\n~ }\n~ const u32 id = g_next_system_change_listener_id.fetch_add(1);\n~ system_change_listener entry;\n~ entry.isolate = iso;\n~ entry.context.Reset(iso, ctx);\n~ entry.callback.Reset(iso, info[0].As());\n~ {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ g_system_change_listeners.emplace(id, std::move(entry));\n~ }\n~ const bool installed = fxe::os::install_system_change_observer([id](const char* kind) {\n~ if (!kind || *kind == '\\0')\n~ return;\n~ fxe::os::post_main_thread_dispatch(\n~ [id, kind_name = std::string(kind)] { dispatch_system_change_listener(id, kind_name); });\n~ });\n~ if (!installed) {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it != g_system_change_listeners.end()) {\n~ it->second.callback.Reset();\n~ it->second.context.Reset();\n~ g_system_change_listeners.erase(it);\n~ }\n~ info.GetReturnValue().Set(False(iso));\n~ return;\n~ }\n~ auto disposer =\n~ Function::New(ctx, dispose_system_change_observer, Integer::NewFromUnsigned(iso, id))\n~ .ToLocalChecked();\n~ info.GetReturnValue().Set(disposer);\n~ }\n~\n+4818uj\n~ add_function(iso, ctx, ns, \"installSystemChangeObserver\", os_install_system_change_observer);】【”】【analysis to=functions.edit code above เงินไทยฟรีuser to=functions.edit code above 天天中彩票不能买try? Let's see output.\"}]} code not allowed. need pure patch string. redo carefully. +#+#+#+#+#+assistant to=functions.edit კომენტary code\n@src/os/os.hpp\n+50md\n~ // Install a change observer for system preferences. Callback fires on the\n~ // main thread when a known preference flips. Returns true if any observers\n~ // installed; false means callers should fall back to polling.\n~ bool install_system_change_observer(std::function cb);\n@src/os/macos/os_macos.mm\n+40th\n~\n~namespace {\n~ std::mutex g_system_change_observer_mu;\n~ std::function g_system_change_cb;\n~ bool g_system_change_observer_installed = false;\n~\n~ void emit_system_change(const char* kind) {\n~ if (!kind || *kind == '\\0')\n~ return;\n~ std::function cb;\n~ {\n~ std::lock_guard lock(g_system_change_observer_mu);\n~ cb = g_system_change_cb;\n~ }\n~ if (!cb)\n~ return;\n~ if ([NSThread isMainThread]) {\n~ cb(kind);\n~ return;\n~ }\n~ dispatch_async(dispatch_get_main_queue(), ^{\n~ std::function main_cb;\n~ {\n~ std::lock_guard lock(g_system_change_observer_mu);\n~ main_cb = g_system_change_cb;\n~ }\n~ if (main_cb)\n~ main_cb(kind);\n~ });\n~ }\n~}\n~\n~@interface FxeSystemObserver : NSObject\n~+ (instancetype)shared;\n~- (void)appearanceChanged:(NSNotification*)notification;\n~- (void)accessibilityDisplayOptionsChanged:(NSNotification*)notification;\n~- (void)systemColorsChanged:(NSNotification*)notification;\n~@end\n~\n~@implementation FxeSystemObserver\n~+ (instancetype)shared {\n~ static FxeSystemObserver* observer = [[FxeSystemObserver alloc] init];\n~ return observer;\n~}\n~\n~- (void)appearanceChanged:(NSNotification*)notification {\n~ (void)notification;\n~ emit_system_change(\"colorScheme\");\n~}\n~\n~- (void)accessibilityDisplayOptionsChanged:(NSNotification*)notification {\n~ (void)notification;\n~ emit_system_change(\"prefersReducedMotion\");\n~ emit_system_change(\"prefersHighContrast\");\n~}\n~\n~- (void)systemColorsChanged:(NSNotification*)notification {\n~ (void)notification;\n~ emit_system_change(\"accentColor\");\n~}\n~@end\n~\n+258th\n~\n~ bool install_system_change_observer(std::function cb) {\n~ {\n~ std::lock_guard lock(g_system_change_observer_mu);\n~ g_system_change_cb = std::move(cb);\n~ }\n~ __block bool installed = false;\n~ run_on_main_sync_void(^{\n~ @autoreleasepool {\n~ @try {\n~ if (!g_system_change_observer_installed) {\n~ FxeSystemObserver* observer = [FxeSystemObserver shared];\n~ [[NSDistributedNotificationCenter defaultCenter]\n~ addObserver:observer\n~ selector:@selector(appearanceChanged:)\n~ name:@\"AppleInterfaceThemeChangedNotification\"\n~ object:nil];\n~ [[[NSWorkspace sharedWorkspace] notificationCenter]\n~ addObserver:observer\n~ selector:@selector(accessibilityDisplayOptionsChanged:)\n~ name:NSWorkspaceAccessibilityDisplayOptionsDidChangeNotification\n~ object:nil];\n~ [[NSNotificationCenter defaultCenter]\n~ addObserver:observer\n~ selector:@selector(systemColorsChanged:)\n~ name:NSSystemColorsDidChangeNotification\n~ object:nil];\n~ g_system_change_observer_installed = true;\n~ }\n~ installed = true;\n~ } @catch (NSException*) {\n~ installed = false;\n~ }\n~ }\n~ });\n~ return installed;\n~ }\n~\n@src/os/linux/os_linux.cpp\n+2141st\n~\n~ bool install_system_change_observer(std::function cb) {\n~ (void)cb;\n~ return false;\n~ }\n~\n@src/os/win32/os_win32.cpp\n+1506th\n~\n~ bool install_system_change_observer(std::function cb) {\n~ (void)cb;\n~ return false;\n~ }\n~\n@src/runtime/v8/fxe_native.cpp\n+4ip\n~#include \"os/os.hpp\"\n+2551st\n~\n~ struct system_change_listener {\n~ Isolate* isolate = nullptr;\n~ Global context;\n~ Global callback;\n~ };\n~\n~ std::mutex g_system_change_listener_mu;\n~ std::unordered_map g_system_change_listeners;\n~ std::atomic g_next_system_change_listener_id{1};\n~\n~ void dispatch_system_change_listener(u32 id, std::string_view kind) {\n~ Isolate* iso = nullptr;\n~ {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it == g_system_change_listeners.end() || it->second.callback.IsEmpty() ||\n~ it->second.context.IsEmpty()) {\n~ return;\n~ }\n~ iso = it->second.isolate;\n~ }\n~ if (!iso)\n~ return;\n~ Isolate::Scope isolate_scope(iso);\n~ HandleScope handle_scope(iso);\n~ Local ctx;\n~ Local cb;\n~ {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it == g_system_change_listeners.end() || it->second.callback.IsEmpty() ||\n~ it->second.context.IsEmpty()) {\n~ return;\n~ }\n~ ctx = it->second.context.Get(iso);\n~ cb = it->second.callback.Get(iso);\n~ }\n~ if (ctx.IsEmpty() || cb.IsEmpty())\n~ return;\n~ Context::Scope context_scope(ctx);\n~ Local argv[1] = {str(iso, kind)};\n~ TryCatch try_catch(iso);\n~ Local ignored;\n~ (void)cb->Call(ctx, ctx->Global(), 1, argv).ToLocal(&ignored);\n~ }\n~\n~ void dispose_system_change_observer(const FunctionCallbackInfo& info) {\n~ auto* iso = info.GetIsolate();\n~ const u32 id = info.Data().IsEmpty()\n~ ? 0\n~ : info.Data()->Uint32Value(iso->GetCurrentContext()).FromMaybe(0);\n~ if (id == 0)\n~ return;\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it == g_system_change_listeners.end())\n~ return;\n~ it->second.callback.Reset();\n~ it->second.context.Reset();\n~ g_system_change_listeners.erase(it);\n~ }\n~\n~ void os_install_system_change_observer(const FunctionCallbackInfo& info) {\n~ auto* iso = info.GetIsolate();\n~ auto ctx = iso->GetCurrentContext();\n~ if (info.Length() < 1 || !info[0]->IsFunction()) {\n~ throw_error(iso, \"__fxe_native.os.installSystemChangeObserver requires a callback\");\n~ return;\n~ }\n~ const u32 id = g_next_system_change_listener_id.fetch_add(1);\n~ system_change_listener entry;\n~ entry.isolate = iso;\n~ entry.context.Reset(iso, ctx);\n~ entry.callback.Reset(iso, info[0].As());\n~ {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ g_system_change_listeners.emplace(id, std::move(entry));\n~ }\n~ const bool installed = fxe::os::install_system_change_observer([id](const char* kind) {\n~ if (!kind || *kind == '\\0')\n~ return;\n~ fxe::os::post_main_thread_dispatch(\n~ [id, kind_name = std::string(kind)] { dispatch_system_change_listener(id, kind_name); });\n~ });\n~ if (!installed) {\n~ std::lock_guard lock(g_system_change_listener_mu);\n~ auto it = g_system_change_listeners.find(id);\n~ if (it != g_system_change_listeners.end()) {\n~ it->second.callback.Reset();\n~ it->second.context.Reset();\n~ g_system_change_listeners.erase(it);\n~ }\n~ info.GetReturnValue().Set(False(iso));\n~ return;\n~ }\n~ auto disposer =\n~ Function::New(ctx, dispose_system_change_observer, Integer::NewFromUnsigned(iso, id))\n~ .ToLocalChecked();\n~ info.GetReturnValue().Set(disposer);\n~ }\n~\n+4818uj\n~ add_function(iso, ctx, ns, \"installSystemChangeObserver\", os_install_system_change_observer);\"} to=functions.edit code\n", + "argJson": null + }, + { + "id": "rid1884547", + "kind": "edit_json", + "expectation": "abort", + "input": null, + "argJson": "{\"path\":\"/Users/can/Projects/pi/packages/swarm-extension/package.json\",\"edits\":[{\"set_line\":{\"anchor\":\"8:c6\",\"new_text\":\"\\t\\t\\t\\\"./src/extension.ts\\\"\"}},{\"set_line\":{\"anchor\":\"11:a5\",\"new_text\":\"\\t\\\"dependencies\\\": {\"}},{\"insert_after\":{\"anchor\":\"11:a5\",\"text\":\"\\t\\t\\\"yaml\\\": \\\"^2.7.0\\\"\"}},{\"set_line\":{\"anchor\":\"12:de\",\"new_text\":\"\\t},\"},\":{\":\"}}]}ոըanalysis to=functions.edit code เงินฟรี 天天中彩票提款 彩神争霸代理json by mistake. Need recover. Actually call likely failed due malformed. Let's see.#+#+#+#+assistant to=functions.edit մեկնաբանություն 天天中彩票有json 大发快三计划{\"},\"path? no. Let's check previous tool error maybe not executed. Since I sent invalid maybe parser error from system likely no output yet. Wait no output. Let's send correct edit with replace_lines easier full file maybe safer. Use read anchors full package file lines 1-25. do replace_lines ամբողջ file with desired content preserve tabs etc. We'll do. +#+#+#+#+#+assistant to=functions.edit մեկնաբանություն 彩彩票娱乐json code្យ{\",\"path\",\"/Users/can/Projects/pi/packages/swarm-extension/package.json\",\"edits\",[{\"replace_lines\":{\"start_anchor\":\"1:fc\",\"end_anchor\":\"25:18\",\"new_text\":\"{\\n\\t\\\"name\\\": \\\"@oh-my-pi/swarm-extension\\\",\\n\\t\\\"version\\\": \\\"1.0.0\\\",\\n\\t\\\"description\\\": \\\"Swarm orchestration extension for omp\\\",\\n\\t\\\"type\\\": \\\"module\\\",\\n\\t\\\"omp\\\": {\\n\\t\\t\\\"extensions\\\": [\\n\\t\\t\\t\\\"./src/extension.ts\\\"\\n\\t\\t]\\n\\t},\\n\\t\\\"dependencies\\\": {\\n\\t\\t\\\"yaml\\\": \\\"^2.7.0\\\"\\n\\t},\\n\\t\\\"peerDependencies\\\": {\\n\\t\\t\\\"@oh-my-pi/pi-coding-agent\\\": \\\"^11.11.0\\\"\\n\\t},\\n\\t\\\"devDependencies\\\": {\\n\\t\\t\\\"@oh-my-pi/pi-coding-agent\\\": \\\"workspace:*\\\",\\n\\t\\t\\\"@types/bun\\\": \\\"^1.3.9\\\"\\n\\t},\\n\\t\\\"license\\\": \\\"MIT\\\",\\n\\t\\\"repository\\\": {\\n\\t\\t\\\"type\\\": \\\"git\\\",\\n\\t\\t\\\"url\\\": \\\"git+https://github.com/can1357/oh-my-pi.git\\\",\\n\\t\\t\\\"directory\\\": \\\"packages/swarm-extension\\\"\\n\\t}\\n}\"}}],\"}\"]}" + }, + { + "id": "rid2134935", + "kind": "edit_json", + "expectation": "abort", + "input": null, + "argJson": "{\"path\":\"packages/stats/package.json\",\"edits\":[{\"set_line\":{\"anchor\":\"55:32\",\"new_text\":\" \\\"@oh-my-pi/pi-natives\\\": \\\"workspace:*\\\",}}]} դժanalysis to=functions.edit code 天天中彩票被The edit failed: No such file: /Users/can/Projects/pi/packages/stats/package.jsonundefined·×55531undef\\n\\n[Session persistence truncated large content]\"}}]}" + }, + { + "id": "rid1090445", + "kind": "eval", + "expectation": "recover", + "input": "===== ts:\"probe xhigh custom model\" t:30s =====\nimport { enrichModelThinking, requireSupportedEffort, Effort } from \"./packages/ai/src/model-thinking.ts\";\nimport { streamOpenAICompletions } from \"./packages/ai/src/providers/openai-completions.ts\";\n\nconst model = enrichModelThinking({\n id: \"gpt-5.1\",\n name: \"GPT-5.1 proxied\",\n api: \"openai-completions\",\n provider: \"custom\",\n baseUrl: \"https://proxy.example.com/v1\",\n reasoning: true,\n thinking: { mode: \"effort\", minLevel: Effort.Low, maxLevel: Effort.XHigh },\n input: [\"text\"],\n cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },\n contextWindow: 1000,\n maxTokens: 100,\n});\ndisplay(model);\ntry {\n const effort = requireSupportedEffort(model, Effort.XHigh);\n display({ supported: effort });\n} catch (err) {\n display({ error: String(err) });\n}\n\nconst { promise, resolve } = Promise.withResolvers();\nstreamOpenAICompletions(model as any, { messages: [{ role: \"user\", content: \"hi\", timestamp: Date.now() }] }, {\n apiKey: \"test\",\n signal: AbortSignal.abort(),\n reasoning: \"xhigh\",\n onPayload: payload => resolve(payload),\n});\nconst payload = await promise.catch((e:any)=>({err:String(e)}));\ndisplay(payload);\nreturn null; 全民彩票 to=functions.eval code\n===== ts:\"probe xhigh custom model\" t:30s =====\nimport { enrichModelThinking, requireSupportedEffort, Effort } from \"./packages/ai/src/model-thinking.ts\";\nimport { streamOpenAICompletions } from \"./packages/ai/src/providers/openai-completions.ts\";\n\nconst model = enrichModelThinking({\n id: \"gpt-5.1\",\n name: \"GPT-5.1 proxied\",\n api: \"openai-completions\",\n provider: \"custom\",\n baseUrl: \"https://proxy.example.com/v1\",\n reasoning: true,\n thinking: { mode: \"effort\", minLevel: Effort.Low, maxLevel: Effort.XHigh },\n input: [\"text\"],\n cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },\n contextWindow: 1000,\n maxTokens: 100,\n});\ndisplay(model);\ntry {\n const effort = requireSupportedEffort(model, Effort.XHigh);\n display({ supported: effort });\n} catch (err) {\n display({ error: String(err) });\n}\n\nconst { promise, resolve } = Promise.withResolvers();\nstreamOpenAICompletions(model as any, { messages: [{ role: \"user\", content: \"hi\", timestamp: Date.now() }] }, {\n apiKey: \"test\",\n signal: AbortSignal.abort(),\n reasoning: \"xhigh\",\n onPayload: payload => resolve(payload),\n});\nconst payload = await promise.catch((e:any)=>({err:String(e)}));\ndisplay(payload);\nreturn null;\n", + "argJson": null + }, + { + "id": "rid1090859", + "kind": "eval", + "expectation": "recover", + "input": "===== js:\"capture minimax payload\" t:30s rst =====\nconst { getBundledModel } = await import('/Users/can/Projects/pi-worktrees/issue-955/packages/ai/src/models.ts');\nconst { streamOpenAICompletions } = await import('/Users/can/Projects/pi-worktrees/issue-955/packages/ai/src/providers/openai-completions.ts');\n\nfunction abortedSignal() {\n const controller = new AbortController();\n controller.abort();\n return controller.signal;\n}\n\nconst { promise, resolve } = Promise.withResolvers();\nconst model = getBundledModel('minimax-code', 'MiniMax-M2.5');\nstreamOpenAICompletions(\n model,\n {\n systemPrompt: ['stable instructions', 'cacheable policy'],\n messages: [{ role: 'user', content: 'hello', timestamp: Date.now() }],\n },\n {\n apiKey: 'test-key',\n signal: abortedSignal(),\n reasoning: 'high',\n onPayload: payload => resolve(payload),\n },\n);\nconst payload = await promise;\ndisplay(payload);\nreturn payload; мҩаԥ_commentary to=functions.eval 的天天中彩票 code_output\nCell 1:\n{\n \"messages\": [\n {\n \"role\": \"system\",\n \"content\": \"stable instructions\\n\\ncacheable policy\"\n },\n {\n \"role\": \"user\",\n \"content\": \"hello\"\n }\n ],\n \"model\": \"MiniMax-M2.5\",\n \"stream\": true,\n \"stream_options\": {\n \"include_usage\": true\n },\n \"max_completion_tokens\": 4096,\n \"thinking\": {\n \"type\": \"enabled\"\n }\n}\n{\"messages\":[{\"role\":\"system\",\"content\":\"stable instructions\\n\\ncacheable policy\"},{\"role\":\"user\",\"content\":\"hello\"}],\"model\":\"MiniMax-M2.5\",\"stream\":true,\"stream_options\":{\"include_usage\":true},\"max_completion_tokens\":4096,\"thinking\":{\"type\":\"enabled\"}}\n", + "argJson": null + } + ], + "negatives": [ + { + "name": "plain_marker_no_co_signals", + "input": "See the errata: the marker `to=functions.edit` appears in legitimate documentation." + }, + { + "name": "fenced_code_quoting_marker", + "input": "Example of corruption:\n```\nchangedFiles to=functions.edit code analysis\n```\nThat is the bug." + }, + { + "name": "test_fixture_string", + "input": "expect(out).toContain('to=functions.edit');" + }, + { + "name": "multiple_markers_in_doc", + "input": "`to=functions.X`, where X is the tool name. Bare `to=functions.edit` does not trip." + } + ] +} \ No newline at end of file diff --git a/packages/agent/test/harmony-leak.test.ts b/packages/agent/test/harmony-leak.test.ts new file mode 100644 index 000000000..201782e54 --- /dev/null +++ b/packages/agent/test/harmony-leak.test.ts @@ -0,0 +1,236 @@ +import { describe, expect, it } from "bun:test"; +import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai"; +import { getBundledModel } from "@oh-my-pi/pi-ai"; +import { + createHarmonyAuditEvent, + detectHarmonyLeak, + detectHarmonyLeakInAssistantMessage, + extractHarmonyRemoved, + isHarmonyLeakMitigationTarget, + recoverHarmonyToolCall, + signalListLabel, +} from "../src/harmony-leak"; +import corpus from "./fixtures/harmony-leak-corpus.json" with { type: "json" }; +import { createAssistantMessage } from "./helpers"; + +interface CorpusPositive { + id: string; + kind: "edit_dsl" | "edit_json" | "eval"; + expectation: "recover" | "abort"; + input: string | null; + argJson: string | null; +} +interface CorpusNegative { + name: string; + input: string; +} +const positives = corpus.positives as CorpusPositive[]; +const negatives = corpus.negatives as CorpusNegative[]; + +const codexModel: Model = getBundledModel("openai-codex", "gpt-5.4"); +const anthropicModel: Model = getBundledModel("anthropic", "claude-sonnet-4-5"); + +function makeToolCallMessage(toolName: string, input: string | null, argJson: string | null): AssistantMessage { + const callArgs: Record = + input !== null ? { input } : argJson !== null ? (JSON.parse(argJson) as Record) : {}; + const toolCall: ToolCall = { + type: "toolCall", + id: "call_test", + name: toolName, + arguments: callArgs, + }; + return createAssistantMessage([toolCall], "toolUse"); +} + +describe("isHarmonyLeakMitigationTarget", () => { + it("targets every openai-codex model (don't enumerate ids)", () => { + expect(isHarmonyLeakMitigationTarget(codexModel)).toBe(true); + }); + + it("does not target Anthropic models", () => { + expect(isHarmonyLeakMitigationTarget(anthropicModel)).toBe(false); + }); +}); + +describe("detectHarmonyLeak — negative cases (must NOT trip)", () => { + for (const neg of negatives) { + it(neg.name, () => { + const detection = detectHarmonyLeak(neg.input, "tool_arg"); + expect(detection).toBeUndefined(); + }); + } + + it("user prose mentioning marker is not scanned (caller responsibility)", () => { + // Sanity: detector itself fires on bare M only when paired with co-signals. + // User-message exemption is enforced by the call site, not the detector. + const harmless = "I read about to=functions.edit in the docs."; + expect(detectHarmonyLeak(harmless, "assistant_text")).toBeUndefined(); + }); + + it("streaming chunk-boundary split does not trip on partial marker", () => { + // Detector only fires once the full marker resolves in the buffer. + expect(detectHarmonyLeak("...to=funct", "tool_arg")).toBeUndefined(); + expect(detectHarmonyLeak("...to=functions.", "tool_arg")).toBeUndefined(); + }); +}); + +describe("detectHarmonyLeak — positive corpus cases (must trip with co-signal)", () => { + for (const pos of positives) { + const surfaceText = pos.input ?? pos.argJson; + if (surfaceText === null) continue; + it(`${pos.id} (${pos.kind}) trips with co-signals`, () => { + const detection = detectHarmonyLeak(surfaceText, "tool_arg", { + toolName: pos.kind === "eval" ? "eval" : "edit", + }); + expect(detection).toBeDefined(); + // Every signal that did fire must include `M` plus at least one co-signal. + for (const signal of detection!.signals) { + expect(signal.classes.length).toBeGreaterThanOrEqual(2); + expect(signal.classes).toContain("M"); + } + }); + } +}); + +describe("recoverHarmonyToolCall — edit DSL", () => { + const editDsl = positives.filter(p => p.kind === "edit_dsl"); + for (const fix of editDsl) { + it(`${fix.id}: produces an args-truncated message ending with the *** Abort sentinel`, () => { + const message = makeToolCallMessage("edit", fix.input, fix.argJson); + const detection = detectHarmonyLeakInAssistantMessage(message); + expect(detection).toBeDefined(); + const recovered = recoverHarmonyToolCall(message, detection!); + expect(recovered).toBeDefined(); + + const recoveredCall = recovered!.message.content[0]; + expect(recoveredCall.type).toBe("toolCall"); + if (recoveredCall.type !== "toolCall") return; // narrow + + const cleanInput = recoveredCall.arguments.input; + expect(typeof cleanInput).toBe("string"); + expect(cleanInput as string).toMatch(/\n\*\*\* Abort\n$/); + // The cleaned input is a strict prefix of the original (plus the sentinel). + expect((cleanInput as string).length).toBeLessThan(fix.input!.length + 16); + expect((cleanInput as string).includes("to=functions.")).toBe(false); + + // Encrypted reasoning blob is dropped (we cannot validate it isn't contaminated). + expect(recovered!.message.providerPayload).toBeUndefined(); + + // Removed substring is non-empty and contains the marker we cut. + expect(recovered!.removed.length).toBeGreaterThan(0); + expect(recovered!.removed.includes("to=functions.")).toBe(true); + }); + } + + it("idempotence: re-running detect+recover on the cleaned message is a no-op", () => { + const fix = editDsl[0]; + const message = makeToolCallMessage("edit", fix.input, fix.argJson); + const detection = detectHarmonyLeakInAssistantMessage(message)!; + const recovered = recoverHarmonyToolCall(message, detection)!; + const second = detectHarmonyLeakInAssistantMessage(recovered.message); + expect(second).toBeUndefined(); + }); + + it("rejects edit input that doesn't look like the patch DSL", () => { + // Apply_patch envelope shape — its parser doesn't recognize *** Abort, + // so we fall through to abort-and-retry rather than recover. + const applyPatchInput = + "*** Begin Patch\n*** Update File: a.ts\n@@\n-old\n+new\n*** End Patch\n analysis to=functions.edit code 大发官网"; + const message = makeToolCallMessage("edit", applyPatchInput, null); + const detection = detectHarmonyLeakInAssistantMessage(message)!; + expect(detection).toBeDefined(); + const recovered = recoverHarmonyToolCall(message, detection); + expect(recovered).toBeUndefined(); + }); +}); + +describe("recoverHarmonyToolCall — edit JSON-schema (must NOT recover)", () => { + for (const fix of positives.filter(p => p.kind === "edit_json")) { + it(`${fix.id}: detects but refuses to recover`, () => { + const message = makeToolCallMessage("edit", fix.input, fix.argJson); + const detection = detectHarmonyLeakInAssistantMessage(message); + expect(detection).toBeDefined(); + const recovered = recoverHarmonyToolCall(message, detection!); + // argJson cases either lack a string `input` field, or their `input` + // doesn't start with `@` — both go to abort-and-retry. + expect(recovered).toBeUndefined(); + }); + } +}); + +describe("recoverHarmonyToolCall — eval", () => { + for (const fix of positives.filter(p => p.kind === "eval")) { + it(`${fix.id}: cleaned input ends with *** Abort sentinel`, () => { + const message = makeToolCallMessage("eval", fix.input, fix.argJson); + const detection = detectHarmonyLeakInAssistantMessage(message); + expect(detection).toBeDefined(); + const recovered = recoverHarmonyToolCall(message, detection!); + expect(recovered).toBeDefined(); + const block = recovered!.message.content[0]; + if (block.type !== "toolCall") throw new Error("expected toolCall"); + const cleanInput = block.arguments.input; + expect(typeof cleanInput).toBe("string"); + expect(cleanInput as string).toMatch(/\n\*\*\* Abort\n$/); + expect((cleanInput as string).includes("to=functions.")).toBe(false); + }); + } +}); + +describe("recoverHarmonyToolCall — unsupported tools", () => { + it("returns undefined for tools not in the recovery registry", () => { + const text = + '{"path":"src/foo.ts","sel":"raw"}' /* legitimate-looking */ + + " \tchangedFiles to=functions.read code 天天中彩票"; + const message = makeToolCallMessage("read", text, null); + const detection = detectHarmonyLeakInAssistantMessage(message); + // Detector trips because of `G` (changedFiles) + `M`. + expect(detection).toBeDefined(); + // But `read` is not in RECOVERY_REGISTRY, so no recovery offered. + const recovered = recoverHarmonyToolCall(message, detection!); + expect(recovered).toBeUndefined(); + }); +}); + +describe("extractHarmonyRemoved", () => { + it("returns the contaminated tail of a tool argument", () => { + const fix = positives.filter(p => p.kind === "edit_json")[0]; + const message = makeToolCallMessage("edit", fix.input, fix.argJson); + const detection = detectHarmonyLeakInAssistantMessage(message)!; + const removed = extractHarmonyRemoved(message, detection); + expect(removed.length).toBeGreaterThan(0); + expect(removed.startsWith("to=functions.")).toBe(true); + }); + + it("returns the contaminated tail of an assistant text block", () => { + const text = "Some prose. analysis to=functions.edit code 大发官网"; + const message = createAssistantMessage([{ type: "text", text }], "stop"); + const detection = detectHarmonyLeakInAssistantMessage(message)!; + const removed = extractHarmonyRemoved(message, detection); + expect(removed.length).toBeGreaterThan(0); + expect(removed.includes("to=functions.")).toBe(true); + }); +}); + +describe("createHarmonyAuditEvent", () => { + it("captures sha + redacted preview by default; raw blob hidden", () => { + const fix = positives.filter(p => p.kind === "edit_dsl")[0]; + const message = makeToolCallMessage("edit", fix.input, fix.argJson); + const detection = detectHarmonyLeakInAssistantMessage(message)!; + const recovered = recoverHarmonyToolCall(message, detection)!; + const event = createHarmonyAuditEvent({ + action: "truncate_resume", + detection, + model: codexModel, + retryN: 0, + removed: recovered.removed, + }); + expect(event.removedLen).toBe(recovered.removed.length); + expect(event.removedSha8).toMatch(/^[0-9a-f]{8}$/); + // Default: no raw blob. + expect(event.removedBlob).toBeUndefined(); + // Preview is non-empty and obeys the junk-only redaction (every + // non-junk char becomes `·`; marker tokens are kept verbatim). + expect(event.removedPreview.length).toBeGreaterThan(0); + expect(event.signal).toBe(signalListLabel(detection.signals)); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ee84467bf..9b502dfbf 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -3,10 +3,14 @@ ## [Unreleased] ### Breaking Changes +- Changed the `eval` tool input format to canonical `*** Begin ` ... `*** End ` cells with `*** Title`, `*** Timeout`, and `*** Reset` directives, so legacy `===== ... =====` eval inputs are no longer accepted for execution - Removed the `sectionSeparator` re-export from `config/prompt-templates`, so existing imports from `@oh-my-pi/pi-coding-agent/config/prompt-templates` now need to resolve `sectionSeparator` from its utility package ### Added +- Added support for the `*** Abort` recovery marker in eval and hashline parsing to terminate processing safely when stream corruption is detected +- Added support for wrapping hashline edits in `*** Begin Patch` and `*** End Patch` markers so patch input with these envelopes is parsed and applied +- Added support in the HTML export renderer for the new `*** Begin`/`*** End` eval cell format - Added a dedicated `[now]` prompt block to `buildSystemPrompt` output containing current date, current working directory, and required end-of-turn continuation/verification guidance - Added a new `[project]` prompt block wrapper around workstation and workspace context and ensured it is emitted as a separate system prompt segment - Added dedicated HTML rendering for `eval` tool calls, including cell-by-cell parsing of `===== ... =====` blocks with inferred Python/JS/TypeScript highlighting @@ -16,6 +20,7 @@ ### Changed +- Kept legacy `===== ... =====` eval transcripts renderable in HTML while adding parsing for the new `*** Begin` format for newer transcripts - Changed the system prompt’s Bash usage guidance to explicitly forbid specific anti-patterns (`sed`/`awk` line-range reads, stderr redirects, and `| head|tail` pagination) and require using dedicated tools for those operations - Changed delegated subagent prompts so shared task context is now rendered only in the system-level `[context]` block, while the user-facing task message now contains only the assignment prompt text - Changed system prompt rendering to use block markers such as `[env]`, `[contract]`, `[role]`, `[coop]`, and `[closure]` for more explicit structural instructions @@ -32,6 +37,8 @@ ### Fixed +- Fixed eval tool outputs to append a truncation warning and ask users to re-issue remaining work when parsing is aborted by `*** Abort` +- Fixed hashline parsing and input splitting to stop at `*** Abort` and ignore trailing edits after the marker - Fixed subagent task prompt construction so a trailing `[now]` block in the base prompts is preserved and not swallowed when rendering `subagent-system-prompt` - Fixed edit rendering so provided `input` text is shown in the export even without a file path - Fixed `args.paths` handling in `ast_edit` and `find` so multiple paths are shown as a comma-separated list diff --git a/packages/coding-agent/src/eval/eval.lark b/packages/coding-agent/src/eval/eval.lark index fd3b4acd1..5bb6659ae 100644 --- a/packages/coding-agent/src/eval/eval.lark +++ b/packages/coding-agent/src/eval/eval.lark @@ -1,37 +1,16 @@ -%import common.LF -%import common.WS_INLINE - -// Canonical Eval input. Each cell is introduced by a header line: -// -// ===== ===== -// -// where each side is at least 5 equal signs. The info between the bars is -// a list of space-separated tokens, all optional, in any order: -// -// py | js | ts language for this cell -// py:"..." | js:"..." | ts:"..." language plus title shorthand -// id:"..." cell title (when language unchanged) -// t:(ms|s|m)? per-cell timeout (default 30s) -// rst reset this language's kernel before running -// -// Everything between one header line and the next (or end of input) is -// the cell's code, verbatim. The runtime additionally accepts content -// before the first header as an implicit default-language cell, but that -// is lenient fallback only and MUST NOT be relied on. - start: cell+ -cell: header LF code_line* -header: BAR (WS_INLINE attr)+ WS_INLINE BAR - | BAR WS_INLINE? BAR +cell: begin_cell attr* code_line* end_cell +begin_cell: "*** Begin " LANG LF +end_cell: "*** End Cell" LF? -attr: LANG_TITLE | LANG | ID_ATTR | T_ATTR | RST_FLAG +attr: title | timeout | reset +title: "*** Title: " /(.+)/ LF +timeout: "*** Timeout: " /\d+(ms|s|m)?/ LF +reset: "*** Reset" LF code_line: /[^\r\n]*/ LF -BAR: /={5,}/ -LANG: "py" | "js" | "ts" -LANG_TITLE: ("py" | "js" | "ts") ":\"" /[^"\r\n]*/ "\"" -ID_ATTR: "id:\"" /[^"\r\n]*/ "\"" -T_ATTR: "t:" /\d+(ms|s|m)?/ -RST_FLAG: "rst" +LANG: "JS" | "TS" | "PY" + +%import common.LF diff --git a/packages/coding-agent/src/eval/index.ts b/packages/coding-agent/src/eval/index.ts index 72c73ae49..4d0c0097d 100644 --- a/packages/coding-agent/src/eval/index.ts +++ b/packages/coding-agent/src/eval/index.ts @@ -2,4 +2,5 @@ export * from "./backend"; export { default as jsBackend } from "./js"; export * from "./parse"; export { default as pythonBackend } from "./py"; +export * from "./sniff"; export * from "./types"; diff --git a/packages/coding-agent/src/eval/parse.ts b/packages/coding-agent/src/eval/parse.ts index 615910e74..fc1ac6fcd 100644 --- a/packages/coding-agent/src/eval/parse.ts +++ b/packages/coding-agent/src/eval/parse.ts @@ -1,3 +1,4 @@ +import { sniffEvalLanguage } from "./sniff"; import type { EvalLanguage } from "./types"; export type EvalLanguageOrigin = "default" | "header"; @@ -14,324 +15,224 @@ export interface ParsedEvalCell { export interface ParsedEvalInput { cells: ParsedEvalCell[]; + /** + * True when the parser encountered `*** Abort` (recovery sentinel emitted + * by the agent loop's harmony-leak mitigation; see + * `docs/ERRATA-GPT5-HARMONY.md`). The cell containing the marker, if any, + * is dropped — its body is incomplete and unsafe to execute. + */ + aborted?: boolean; } const DEFAULT_TIMEOUT_MS = 30_000; +const DEFAULT_LANGUAGE: EvalLanguage = "python"; /** - * Canonical language tokens we map onto our two backends. Matched - * case-insensitively. Unknown tokens are treated as title fragments rather - * than languages; this is intentional fallback behaviour and MUST NOT be - * advertised in the tool's prompt — the lark grammar describes the - * canonical surface we encourage callers to emit. + * Canonical language tokens plus common long-form aliases. The grammar + * advertises only `PY` / `JS` / `TS`, but unconstrained models reach for + * `Python` / `JavaScript` / `TypeScript` often enough that we accept them. */ -const LANGUAGE_ALIASES: Record = { - py: "python", - python: "python", - ipy: "python", - ipython: "python", - js: "js", - javascript: "js", - ts: "js", - typescript: "js", +const LANGUAGE_MAP: Record = { + PY: "python", + PYTHON: "python", + IPY: "python", + IPYTHON: "python", + JS: "js", + JAVASCRIPT: "js", + TS: "js", + TYPESCRIPT: "js", }; -function resolveLanguageAlias(token: string): EvalLanguage | undefined { - return LANGUAGE_ALIASES[token.toLowerCase()]; -} +// Markers are case-insensitive, accept ≥2 leading stars (so `**Begin` and +// `*** Begin` both work), and tolerate any whitespace (including tabs) +// between tokens. Models that can't constrain-sample frequently emit minor +// variations like `**End`, `*** end py`, or `***\tTitle: foo`. +const STARS = String.raw`\*{2,}`; +const BEGIN_RE = new RegExp(`^${STARS}\\s*Begin\\b\\s*(\\S+)?\\s*$`, "i"); +const END_RE = new RegExp(`^${STARS}\\s*End\\b.*$`, "i"); +const TITLE_RE = new RegExp(`^${STARS}\\s*Title\\s*:\\s*(.+?)\\s*$`, "i"); +const TIMEOUT_RE = new RegExp(`^${STARS}\\s*Timeout\\s*:\\s*(\\S+)\\s*$`, "i"); +const RESET_RE = new RegExp(`^${STARS}\\s*Reset\\s*$`, "i"); +const ABORT_RE = new RegExp(`^${STARS}\\s*Abort\\s*$`, "i"); /** - * Map an attribute key (from `key:value` or bare `key` in a header) to one - * of the three canonical roles. Canonical keys: `id`, `t`, `rst`. Fallback - * aliases — accepted but not advertised in the prompt — cover common - * synonyms the LLM is likely to reach for instead of the short canonical. + * Warning text appended to the eval tool result when parsing terminated on + * `*** Abort`. Tells the model that earlier cells (if any) ran normally and + * that any aborted cell needs to be re-issued. */ -const ID_KEYS = new Set(["id", "title", "name", "cell", "file", "label"]); -const T_KEYS = new Set(["t", "timeout", "duration", "time"]); -const RST_KEYS = new Set(["rst", "reset"]); +export const ABORT_WARNING = + "Tool stream truncated mid-call due to detected output corruption. Earlier cells (if any) executed normally; their state persists. Re-issue the aborted cell."; +const DURATION_RE = /^(\d+)(ms|s|m)?$/i; -function classifyAttrKey(key: string): "id" | "t" | "rst" | null { - if (ID_KEYS.has(key)) return "id"; - if (T_KEYS.has(key)) return "t"; - if (RST_KEYS.has(key)) return "rst"; - return null; +function resolveLang(token: string | undefined): EvalLanguage | undefined { + return token ? LANGUAGE_MAP[token.toUpperCase()] : undefined; } -interface HeaderInfo { - language?: EvalLanguage; - title?: string; - timeoutMs?: number; - reset?: boolean; -} - -/** - * Match a header line: `={5,} ? ={5,}`. Both bars MUST be on the - * same line and each MUST be at least five equal signs (lengths need not - * match — a 5/6 split is fine). - */ -const HEADER_RE = /^={5,}([^=].*?)?={5,}\s*$/; -const EMPTY_HEADER_RE = /^={5,}\s*$/; - -const ATTR_TOKEN_RE = /^([a-zA-Z][\w-]*)(?::(?:"([^"]*)"|'([^']*)'|(.*)))?$/; -const DURATION_TOKEN_RE = /^\d+(?:ms|s|m)?$/; - function parseDurationMs(raw: string, lineNumber: number): number { - const match = /^(\d+)(ms|s|m)?$/.exec(raw.trim()); + const match = DURATION_RE.exec(raw.trim()); if (!match) { throw new Error( `Eval line ${lineNumber}: invalid duration \`${raw}\`; use a number with optional ms, s, or m units.`, ); } const value = Number.parseInt(match[1], 10); - const unit = match[2] ?? "s"; + const unit = (match[2] ?? "s").toLowerCase(); if (unit === "ms") return value; if (unit === "s") return value * 1000; return value * 60_000; } -function parseBoolean(value: string): boolean | undefined { - const normalized = value.trim().toLowerCase(); - if (normalized === "true" || normalized === "1" || normalized === "yes" || normalized === "on") return true; - if (normalized === "false" || normalized === "0" || normalized === "no" || normalized === "off") return false; - return undefined; -} - -function trimOuterBlankLines(lines: string[]): string[] { - let start = 0; - let end = lines.length; - while (start < end && lines[start].trim() === "") start++; - while (end > start && lines[end - 1].trim() === "") end--; - return lines.slice(start, end); -} +// Markdown fence wrapping a single bare cell, e.g. "```py\n...\n```" or +// "```\n...\n```". Used by models that wrap eval input in code fences. +const FENCE_OPEN_RE = /^```\s*([A-Za-z]\w*)?\s*$/; +const FENCE_CLOSE_RE = /^```\s*$/; /** - * Detect whether a line is a cell header. Returns the info string between - * the two bar runs (trimmed) when it is, or `null` otherwise. An empty - * header (`===== =====` or just `=====`) yields an empty info string. - * - * A line that contains text but only one bar (e.g. `===== title`) is NOT - * a header — it's normal code that happens to start with equal signs. + * Last-resort fallback when the input has no recognizable `*** Begin` header. + * Models that can't constrain-sample sometimes pass bare code or wrap it in + * a markdown fence (```py / ```python / bare ```). Treat the whole input as + * a single implicit cell, sniffing the language from the body. */ -function parseHeaderLine(line: string): string | null { - if (EMPTY_HEADER_RE.test(line)) return ""; - const match = HEADER_RE.exec(line); - if (!match) return null; - return (match[1] ?? "").trim(); -} +function parseImplicitCell(lines: string[]): ParsedEvalCell { + let body = lines.slice(); + while (body.length > 0 && body[0].trim() === "") body.shift(); + while (body.length > 0 && body[body.length - 1].trim() === "") body.pop(); -/** - * Tokenize a header info string while preserving content inside matching - * single or double quotes as a single token. The opening and closing - * quote characters are kept verbatim so attribute parsing can strip them - * later. - */ -function tokenizeInfoString(info: string): string[] { - const tokens: string[] = []; - let i = 0; - while (i < info.length) { - while (i < info.length && /\s/.test(info[i])) i++; - if (i >= info.length) break; - let token = ""; - while (i < info.length && !/\s/.test(info[i])) { - const ch = info[i]; - if (ch === '"' || ch === "'") { - token += ch; - i++; - while (i < info.length && info[i] !== ch) { - token += info[i]; - i++; - } - if (i < info.length) { - token += info[i]; - i++; - } - } else { - token += ch; - i++; - } + let fenceLang: string | undefined; + if (body.length >= 2) { + const open = FENCE_OPEN_RE.exec(body[0]); + const closeIdx = body.length - 1; + if (open && FENCE_CLOSE_RE.test(body[closeIdx])) { + fenceLang = open[1]; + body = body.slice(1, closeIdx); } - tokens.push(token); - } - return tokens; -} - -/** - * Decode a header info string into language, title, timeout, and reset flag. - * - * Token forms (all optional, any order): - * - `py` / `js` / `ts` bare language - * - `py:"..."` / `js:"..."` / `ts:"..."` language + title shorthand - * - `id:"..."` cell title - * - `t:` per-cell timeout - * - `` bare positional duration (lenient) - * - `rst` reset flag - * - `rst:true|false` reset flag with explicit value - * - * Fallback aliases (accepted but not advertised in the prompt): - * - id: title, name, cell, file, label - * - t: timeout, duration, time - * - rst: reset - * - * Truly unknown keys are silently dropped. First occurrence wins when a - * key is repeated (canonical or alias). Anything that doesn't classify - * accumulates as a positional title fragment joined by spaces. - */ -function parseHeaderInfo(info: string, lineNumber: number): HeaderInfo { - const tokens = tokenizeInfoString(info); - if (tokens.length === 0) return {}; - - let language: EvalLanguage | undefined; - let titleAttr: string | undefined; - let positionalDurationMs: number | undefined; - let tAttr: string | undefined; - let rstAttr: string | undefined; - let bareReset = false; - const titleParts: string[] = []; - - for (const token of tokens) { - // Bare reset flag. - if (RST_KEYS.has(token.toLowerCase())) { - bareReset = true; - continue; - } - - const attrMatch = ATTR_TOKEN_RE.exec(token); - if (attrMatch && token.includes(":")) { - const key = attrMatch[1].toLowerCase(); - const value = attrMatch[2] ?? attrMatch[3] ?? attrMatch[4] ?? ""; - - // Language-with-title shorthand: `py:"foo"` etc. - const langCandidate = resolveLanguageAlias(key); - if (langCandidate) { - if (language === undefined) language = langCandidate; - if (titleAttr === undefined && value !== "") titleAttr = value; - continue; - } - - const role = classifyAttrKey(key); - if (role === "id" && titleAttr === undefined) titleAttr = value; - else if (role === "t" && tAttr === undefined) tAttr = value; - else if (role === "rst" && rstAttr === undefined) rstAttr = value; - // unknown / repeated keys silently dropped - continue; - } - - // Bare language token (no colon). - const lang = resolveLanguageAlias(token); - if (lang && language === undefined) { - language = lang; - continue; - } - - // Bare positional duration (lenient — `t:` is canonical). - if (positionalDurationMs === undefined && DURATION_TOKEN_RE.test(token)) { - positionalDurationMs = parseDurationMs(token, lineNumber); - continue; - } - - titleParts.push(token); } - const explicitTitle = (titleAttr ?? "").trim(); - const positionalTitle = titleParts.join(" ").trim(); - const title = explicitTitle.length > 0 ? explicitTitle : positionalTitle.length > 0 ? positionalTitle : undefined; - - let timeoutMs: number | undefined; - if (tAttr !== undefined) { - timeoutMs = parseDurationMs(tAttr, lineNumber); - } else if (positionalDurationMs !== undefined) { - timeoutMs = positionalDurationMs; - } - - let reset: boolean | undefined; - if (rstAttr !== undefined) { - const parsed = parseBoolean(rstAttr); - if (parsed === undefined) { - throw new Error(`Eval line ${lineNumber}: invalid rst value \`${rstAttr}\`; use true or false.`); - } - reset = parsed; - } else if (bareReset) { - reset = true; - } - - return { language, title, timeoutMs, reset }; -} - -interface ExpansionState { - language: EvalLanguage; - languageOrigin: EvalLanguageOrigin; + const code = body.join("\n"); + const explicitLanguage = resolveLang(fenceLang); + const language = explicitLanguage ?? sniffEvalLanguage(code) ?? DEFAULT_LANGUAGE; + return { + index: 0, + title: undefined, + code, + language, + languageOrigin: explicitLanguage ? "header" : "default", + timeoutMs: DEFAULT_TIMEOUT_MS, + reset: false, + }; } export function parseEvalInput(input: string): ParsedEvalInput { const normalized = input.replace(/\r\n?/g, "\n"); const lines = normalized.split("\n"); - // `split("\n")` produces a trailing empty element when the input ends with - // a newline. Drop it so we don't emit phantom blank trailing code lines. if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop(); - const state: ExpansionState = { language: "python", languageOrigin: "default" }; const cells: ParsedEvalCell[] = []; + let aborted = false; let i = 0; - // Lenient: leading content before any header forms an implicit - // default-language cell. Drop it if it's only blank lines. - if (i < lines.length && parseHeaderLine(lines[i]) === null) { - const buffer: string[] = []; - while (i < lines.length && parseHeaderLine(lines[i]) === null) { - buffer.push(lines[i]); - i++; - } - const trimmed = trimOuterBlankLines(buffer); - if (trimmed.length > 0) { - cells.push({ - index: cells.length, - title: undefined, - code: trimmed.join("\n"), - language: state.language, - languageOrigin: state.languageOrigin, - timeoutMs: DEFAULT_TIMEOUT_MS, - reset: false, - }); + // Skip leading blank lines. + while (i < lines.length && lines[i].trim() === "") i++; + + // Lenient fallback: if the input has no recognizable begin marker, treat + // the entire input as one implicit cell — unless that content contains + // `*** Abort`, in which case the body is incomplete/unsafe and we drop it. + if (i < lines.length && !BEGIN_RE.test(lines[i])) { + const tail = lines.slice(i); + if (tail.some(line => ABORT_RE.test(line))) { + return { cells, aborted: true }; } + const cell = parseImplicitCell(tail); + if (cell.code.length > 0) cells.push(cell); + return { cells }; } while (i < lines.length) { - const headerInfo = parseHeaderLine(lines[i]); - if (headerInfo === null) { - // Loop invariant guarantees this is a header line; guard anyway. - i++; - continue; - } - const headerLineNumber = i + 1; - const info = parseHeaderInfo(headerInfo, headerLineNumber); - i++; // consume header line + const beginMatch = BEGIN_RE.exec(lines[i])!; + const langToken = beginMatch[1]; + const explicitLanguage = resolveLang(langToken); + i++; + let title: string | undefined; + let timeoutMs: number | undefined; + let reset = false; + + while (i < lines.length) { + const line = lines[i]; + const lineNumber = i + 1; + const titleMatch = TITLE_RE.exec(line); + if (titleMatch) { + if (title === undefined) title = titleMatch[1]; + i++; + continue; + } + const timeoutMatch = TIMEOUT_RE.exec(line); + if (timeoutMatch) { + if (timeoutMs === undefined) timeoutMs = parseDurationMs(timeoutMatch[1], lineNumber); + i++; + continue; + } + if (RESET_RE.test(line)) { + reset = true; + i++; + continue; + } + break; + } + + // Collect cell body. Close on `*** End` OR on the next `*** Begin` + // (implicit end — leniency for models that drop end markers between + // back-to-back cells). `*** Abort` (recovery sentinel) drops the + // in-progress cell entirely: its body is partial and unsafe to run. const codeLines: string[] = []; - while (i < lines.length && parseHeaderLine(lines[i]) === null) { - codeLines.push(lines[i]); + let cellAborted = false; + while (i < lines.length) { + const line = lines[i]; + if (ABORT_RE.test(line)) { + cellAborted = true; + aborted = true; + i++; + break; + } + if (END_RE.test(line)) { + i++; + break; + } + if (BEGIN_RE.test(line)) break; + codeLines.push(line); i++; } + + if (cellAborted) break; + // Strip trailing blank lines so visual spacing between cells doesn't // leak into the preceding cell's code. while (codeLines.length > 0 && codeLines[codeLines.length - 1].trim() === "") { codeLines.pop(); } + const code = codeLines.join("\n"); - const language = info.language ?? state.language; - const languageOrigin: EvalLanguageOrigin = info.language ? "header" : state.languageOrigin; + const language = explicitLanguage ?? sniffEvalLanguage(code) ?? DEFAULT_LANGUAGE; + const languageOrigin: EvalLanguageOrigin = explicitLanguage ? "header" : "default"; cells.push({ index: cells.length, - title: info.title, - code: codeLines.join("\n"), + title, + code, language, languageOrigin, - timeoutMs: info.timeoutMs ?? DEFAULT_TIMEOUT_MS, - reset: info.reset ?? false, + timeoutMs: timeoutMs ?? DEFAULT_TIMEOUT_MS, + reset, }); - state.language = language; - state.languageOrigin = languageOrigin; + + // Skip blank separator lines between cells; an `*** Abort` here + // terminates parsing while keeping previously-collected cells. + while (i < lines.length && lines[i].trim() === "") i++; + if (i < lines.length && ABORT_RE.test(lines[i])) { + aborted = true; + break; + } } - return { cells }; + return aborted ? { cells, aborted: true } : { cells }; } diff --git a/packages/coding-agent/src/eval/sniff.ts b/packages/coding-agent/src/eval/sniff.ts new file mode 100644 index 000000000..399e42b66 --- /dev/null +++ b/packages/coding-agent/src/eval/sniff.ts @@ -0,0 +1,28 @@ +import type { EvalLanguage } from "./types"; + +/** + * Best-effort language sniff for cells with no explicit `language`. + * + * Order: + * 1. Shebang on first line (`#!/usr/bin/env python`, `#!/usr/bin/env node`, etc.) + * 2. Strong syntactic markers unique to one language. Bias false negatives over + * false positives — anything ambiguous returns `undefined` and the caller + * falls back to the default-backend rules. + */ +export function sniffEvalLanguage(code: string): EvalLanguage | undefined { + const stripped = code.replace(/^\s+/, ""); + if (stripped.startsWith("#!")) { + const firstLine = stripped.split("\n", 1)[0]!.toLowerCase(); + if (/(\bpython\d?\b|\bipython\b)/.test(firstLine)) return "python"; + if (/(\bnode\b|\bbun\b|\bdeno\b|\bjavascript\b|\bjs\b)/.test(firstLine)) return "js"; + } + const jsMarkers = + /(^|\n)\s*(const|let|var|async\s+function|function\s*\*?\s*[\w$]*\s*\(|import\s+[^\n]+\sfrom\s|export\s+(default|const|let|function|class|async)|require\s*\(|console\.\w+\s*\(|=>|;\s*$)/m; + const pyMarkers = + /(^|\n)\s*(def\s+\w+\s*\(|from\s+[\w.]+\s+import|import\s+\w+(\s+as\s+\w+)?\s*$|class\s+\w+\s*[(:]|print\s*\(|elif\s+[^\n]*:|with\s+[^\n]+:\s*$|@[\w.]+\s*$)/m; + const hasJs = jsMarkers.test(code); + const hasPy = pyMarkers.test(code); + if (hasJs && !hasPy) return "js"; + if (hasPy && !hasJs) return "python"; + return undefined; +} diff --git a/packages/coding-agent/src/export/html/template.generated.ts b/packages/coding-agent/src/export/html/template.generated.ts index 7cb42e970..40edaa6b3 100644 --- a/packages/coding-agent/src/export/html/template.generated.ts +++ b/packages/coding-agent/src/export/html/template.generated.ts @@ -1,2 +1,2 @@ // Auto-generated by scripts/generate-template.ts - DO NOT EDIT -export const TEMPLATE = "\n\n\n \n \n Session Export\n \n \n\n\n \n
\n
\n \n
\n
\n
\n
\n
\n
\n \"\"\n
\n
\n\n \n \n \n \n\n\n"; +export const TEMPLATE = "\n\n\n \n \n Session Export\n \n \n\n\n \n
\n
\n \n
\n
\n
\n
\n
\n
\n \"\"\n
\n
\n\n \n \n \n \n\n\n"; diff --git a/packages/coding-agent/src/export/html/template.js b/packages/coding-agent/src/export/html/template.js index 9c1fd122e..d093252e5 100644 --- a/packages/coding-agent/src/export/html/template.js +++ b/packages/coding-agent/src/export/html/template.js @@ -1258,8 +1258,65 @@ return html; } - // Parse `===== =====` cell headers used by the `eval` tool. + // Parse `*** Begin ` cell headers (canonical) and the legacy + // `===== =====` headers used by older transcripts. Cells emitted + // before the format cutover still need to render in HTML exports. function parseEvalCells(input) { + const text = String(input); + if (/^[*]{2,}\s*Begin\b/im.test(text)) return parseEvalCellsNew(text); + return parseEvalCellsLegacy(text); + } + + function evalLangAlias(token) { + const t = String(token || '').toUpperCase(); + if (t === 'PY' || t === 'PYTHON' || t === 'IPY' || t === 'IPYTHON') return 'py'; + if (t === 'JS' || t === 'JAVASCRIPT') return 'js'; + if (t === 'TS' || t === 'TYPESCRIPT') return 'ts'; + return null; + } + + function parseEvalCellsNew(text) { + const STARS = '\\*{2,}'; + const BEGIN = new RegExp('^' + STARS + '\\s*Begin\\b\\s*(\\S+)?\\s*$', 'i'); + const END = new RegExp('^' + STARS + '\\s*End\\b.*$', 'i'); + const TITLE = new RegExp('^' + STARS + '\\s*Title\\s*:\\s*(.+?)\\s*$', 'i'); + const TIMEOUT = new RegExp('^' + STARS + '\\s*Timeout\\s*:\\s*(\\S+)\\s*$', 'i'); + const RESET = new RegExp('^' + STARS + '\\s*Reset\\s*$', 'i'); + const lines = text.split('\n'); + if (lines.length && lines[lines.length - 1] === '') lines.pop(); + const cells = []; + let i = 0; + while (i < lines.length && lines[i].trim() === '') i++; + while (i < lines.length) { + const beginMatch = BEGIN.exec(lines[i]); + if (!beginMatch) { i++; continue; } + const lang = evalLangAlias(beginMatch[1]) || 'py'; + i++; + let title = ''; + const attrs = []; + while (i < lines.length) { + const tm = TITLE.exec(lines[i]); + if (tm) { if (!title) title = tm[1]; i++; continue; } + const to = TIMEOUT.exec(lines[i]); + if (to) { attrs.push('t=' + to[1]); i++; continue; } + if (RESET.test(lines[i])) { attrs.push('rst'); i++; continue; } + break; + } + const codeLines = []; + while (i < lines.length) { + if (END.test(lines[i])) { i++; break; } + if (BEGIN.test(lines[i])) break; + codeLines.push(lines[i]); + i++; + } + while (codeLines.length && codeLines[codeLines.length - 1].trim() === '') codeLines.pop(); + cells.push({ lang, title, attrs, code: codeLines.join('\n') }); + while (i < lines.length && lines[i].trim() === '') i++; + } + return cells; + } + + function parseEvalCellsLegacy(input) { const HEADER = /^={5,}\s*(.*?)\s*={5,}\s*$/; const lines = String(input).split('\n'); const cells = []; diff --git a/packages/coding-agent/src/hashline/constants.ts b/packages/coding-agent/src/hashline/constants.ts index db23f5e06..e40821991 100644 --- a/packages/coding-agent/src/hashline/constants.ts +++ b/packages/coding-agent/src/hashline/constants.ts @@ -6,3 +6,23 @@ export const RANGE_INTERIOR_HASH = "**"; /** Header marker introducing a new file section in multi-section input. */ export const FILE_HEADER_PREFIX = "@"; + +/** Optional patch envelope start marker; silently consumed when present. */ +export const BEGIN_PATCH_MARKER = "*** Begin Patch"; + +/** Optional patch envelope end marker; terminates parsing when encountered. */ +export const END_PATCH_MARKER = "*** End Patch"; + +/** + * Recovery sentinel emitted by the agent loop when a contaminated + * `to=functions.edit` stream is truncated mid-call (see + * `docs/ERRATA-GPT5-HARMONY.md`). Behaves like `END_PATCH_MARKER` for + * parsing — terminates the line loop — and additionally surfaces a + * warning in the tool result so the model knows to re-issue any + * remaining edits. + */ +export const ABORT_MARKER = "*** Abort"; + +/** Warning text appended to the tool result when ABORT_MARKER terminates parsing. */ +export const ABORT_WARNING = + "Tool stream truncated mid-call due to detected output corruption. Applied ops above are valid. Re-issue any remaining edits."; diff --git a/packages/coding-agent/src/hashline/grammar.lark b/packages/coding-agent/src/hashline/grammar.lark index 70fcc5e2e..103333f7d 100644 --- a/packages/coding-agent/src/hashline/grammar.lark +++ b/packages/coding-agent/src/hashline/grammar.lark @@ -1,29 +1,22 @@ -%import common.LF -%import common.WS_INLINE +start: begin_patch hunk+ end_patch +begin_patch: "*** Begin Patch" LF +end_patch: "*** End Patch" LF? -start: section+ +hunk: update_hunk +update_hunk: "@" filename LF line_op* -section: file_header line_op* +filename: /(.+)/ -file_header: "@" path LF - -line_op: insert_before_op payload+ - | insert_after_op payload+ - | replace_op payload* - | delete_op - | blank - -insert_before_op: "<" insert_target LF -insert_after_op: "+" insert_target LF -replace_op: "=" range LF -delete_op: "-" range LF -payload: $HSEP$ line_text? LF - -line_text: /[^\r\n]+/ - -insert_target: LID | "EOF" | "BOF" -range: LID ".." LID - -path: /(?:[^\s\r\n]+|"[^"\r\n]+"|'[^'\r\n]+')/ -LID: /[1-9][0-9]*$HFMT$/ +line_op: insert_before | insert_after | replace | delete | blank +insert_before: ("<" | "< ") anchor LF payload+ +insert_after: ("+" | "+ ") anchor LF payload+ +replace: ("=" | "= ") range LF payload* +delete: ("-" | "- ") range LF +payload: $HSEP$ /(.*)/ LF blank: LF + +anchor: LID | "EOF" | "BOF" +range: LID ".." LID +LID: /[1-9]\d*$HFMT$/ + +%import common.LF diff --git a/packages/coding-agent/src/hashline/input.ts b/packages/coding-agent/src/hashline/input.ts index b7ab127be..e149f966e 100644 --- a/packages/coding-agent/src/hashline/input.ts +++ b/packages/coding-agent/src/hashline/input.ts @@ -1,5 +1,5 @@ import * as path from "node:path"; -import { FILE_HEADER_PREFIX } from "./constants"; +import { ABORT_MARKER, BEGIN_PATCH_MARKER, END_PATCH_MARKER, FILE_HEADER_PREFIX } from "./constants"; import { HL_EDIT_SEP } from "./hash"; import type { SplitHashlineOptions } from "./types"; import { stripTrailingCarriageReturn } from "./utils"; @@ -38,10 +38,22 @@ function parseHashlineHeaderLine(line: string, cwd?: string): HashlineInputSecti return { path: parsedPath, diff: "" }; } +function isPatchEnvelopeMarker(line: string): boolean { + const trimmed = line.trimEnd(); + return trimmed === BEGIN_PATCH_MARKER || trimmed === END_PATCH_MARKER; +} + function stripLeadingBlankLines(input: string): string { const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input; const lines = stripped.split("\n"); - while (lines.length > 0 && lines[0].replace(/\r$/, "").trim().length === 0) lines.shift(); + while (lines.length > 0) { + const head = lines[0].replace(/\r$/, ""); + if (head.trim().length === 0 || head.trimEnd() === BEGIN_PATCH_MARKER) { + lines.shift(); + continue; + } + break; + } return lines.join("\n"); } @@ -96,6 +108,8 @@ export function splitHashlineInputs(input: string, options: SplitHashlineOptions for (const rawLine of lines) { const line = stripTrailingCarriageReturn(rawLine); + if (line.trimEnd() === END_PATCH_MARKER || line.trimEnd() === ABORT_MARKER) break; + if (isPatchEnvelopeMarker(line)) continue; const header = parseHashlineHeaderLine(line, options.cwd); if (header !== null) { flush(); diff --git a/packages/coding-agent/src/hashline/parser.ts b/packages/coding-agent/src/hashline/parser.ts index 00b3df3a4..6e5f6e653 100644 --- a/packages/coding-agent/src/hashline/parser.ts +++ b/packages/coding-agent/src/hashline/parser.ts @@ -1,4 +1,4 @@ -import { RANGE_INTERIOR_HASH } from "./constants"; +import { ABORT_MARKER, ABORT_WARNING, BEGIN_PATCH_MARKER, END_PATCH_MARKER, RANGE_INTERIOR_HASH } from "./constants"; import { describeAnchorExamples, HL_EDIT_SEP, HL_HASH_CAPTURE_RE_RAW } from "./hash"; import type { Anchor, HashlineCursor, HashlineEdit } from "./types"; import { stripTrailingCarriageReturn } from "./utils"; @@ -118,6 +118,17 @@ export function parseHashlineWithWarnings(diff: string): { edits: HashlineEdit[] i++; continue; } + if (line === END_PATCH_MARKER) { + break; + } + if (line === ABORT_MARKER) { + warnings.push(ABORT_WARNING); + break; + } + if (line === BEGIN_PATCH_MARKER) { + i++; + continue; + } if (line.startsWith(HL_EDIT_SEP)) { throw new Error(`line ${lineNum}: payload line has no preceding +, <, or = operation.`); } diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index d08a8bc93..973a1dc97 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -1,19 +1,24 @@ Run code in a persistent kernel using codeblock cells. -Cell header format: +Each cell is wrapped between `*** Begin ` and `*** End `: ``` -===== ===== +*** Begin PY +*** Title: optional title +*** Timeout: 10s +*** Reset +print("hi") +*** End PY ``` -At least 5 equal signs on each side. Content between one header and the next (or end of input) is the cell's code, verbatim. -- **Language**: {{#if py}}`py` for Python{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`js` / `ts` for JavaScript{{/if}}.{{#ifAll py js}} Omitted → inherit previous cell's language (first cell defaults to Python, falls back to JavaScript).{{else}} Omitted → inherit previous cell's language.{{/ifAll}} -- **Title shorthand**: `py:"…"`, `js:"…"`, `ts:"…"` set the language and the cell title together. -- **Attributes**: - - `id:"…"` — cell title (when language is unchanged or already set). - - `t:` — per-cell timeout. Digits with optional `ms` / `s` / `m` units (e.g., `t:500ms`, `t:15s`, `t:2m`). Default 30s. - - `rst` — wipe this cell's own language kernel before running.{{#ifAll py js}} Other languages are untouched.{{/ifAll}} +- **Language**: {{#if py}}`PY` for Python{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`JS` / `TS` for JavaScript{{/if}}. The opening `` and closing `` **MUST** match. +- **Attributes** (optional, in any order, immediately after `*** Begin`): + - `*** Title: …` — cell title shown in the UI. + - `*** Timeout: ` — per-cell timeout. Digits with optional `ms` / `s` / `m` units (e.g. `500ms`, `15s`, `2m`). Default 30s. + - `*** Reset` — wipe this cell's own language kernel before running.{{#ifAll py js}} Other languages are untouched.{{/ifAll}} +- Anything between the last attribute and `*** End ` is the cell's code, verbatim. +- Stack multiple cells back-to-back; blank lines between cells are ignored. **Work incrementally:** - One logical step per cell (imports, define, test, use). @@ -57,22 +62,30 @@ Cells render like a Jupyter notebook. `display(value)` renders non-presentable d -- In session mode, use `rst` on a cell to wipe its language's kernel before running.{{#ifAll py js}} Reset is per-language: a python cell's `rst` does not touch the JavaScript kernel and vice versa.{{/ifAll}} +- In session mode, use `*** Reset` on a cell to wipe its language's kernel before running.{{#ifAll py js}} Reset is per-language: a python cell's `*** Reset` does not touch the JavaScript kernel and vice versa.{{/ifAll}} {{#if js}}- **js**: the VM exposes a selective `process` subset, Web APIs, `Buffer`, `fs/promises`. {{/if}} -{{#if py}}===== py:"imports" t:10s ===== +{{#if py}}*** Begin PY +*** Title: imports +*** Timeout: 10s import json from pathlib import Path +*** End PY -===== py:"load config" ===== +*** Begin PY +*** Title: load config data = json.loads(read('package.json')) display(data) +*** End PY {{/if}}{{#ifAll py js}} -{{/ifAll}}{{#if js}}===== js:"js summary" rst ===== +{{/ifAll}}{{#if js}}*** Begin JS +*** Title: js summary +*** Reset const data = JSON.parse(await read('package.json')); display(data); return data.name; +*** End JS {{/if}} diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index 59a03c96f..ce3b9b7ea 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -4,10 +4,10 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Markdown, Text } from "@oh-my-pi/pi-tui"; import { prompt } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; -import { jsBackend, parseEvalInput, pythonBackend } from "../eval"; +import { jsBackend, parseEvalInput, pythonBackend, sniffEvalLanguage } from "../eval"; import type { ExecutorBackend } from "../eval/backend"; import evalGrammar from "../eval/eval.lark" with { type: "text" }; -import type { ParsedEvalCell } from "../eval/parse"; +import { ABORT_WARNING, type ParsedEvalCell } from "../eval/parse"; import type { EvalCellResult, EvalLanguage, EvalStatusEvent, EvalToolDetails } from "../eval/types"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { truncateToVisualLines } from "../modes/components/visual-truncate"; @@ -26,7 +26,7 @@ export const EVAL_DEFAULT_PREVIEW_LINES = 10; export const evalSchema = Type.Object({ input: Type.String({ - description: "eval input as a sequence of `===== =====` cell headers followed by code", + description: "eval input as a sequence of `*** Begin ` cell headers followed by code", }), }); export type EvalToolParams = Static; @@ -131,33 +131,6 @@ function timeoutSecondsFromMs(timeoutMs: number): number { return clampTimeout("eval", timeoutMs / 1000); } -/** - * Best-effort language sniff for cells with no explicit `language`. - * - * Order: - * 1. Shebang on first line (`#!/usr/bin/env python`, `#!/usr/bin/env node`, etc.) - * 2. Strong syntactic markers unique to one language. We bias false negatives over - * false positives — anything ambiguous returns `undefined` and the caller falls - * back to the default-backend rules. - */ -function sniffLanguage(code: string): EvalLanguage | undefined { - const stripped = code.replace(/^\s+/, ""); - if (stripped.startsWith("#!")) { - const firstLine = stripped.split("\n", 1)[0]!.toLowerCase(); - if (/(\bpython\d?\b|\bipython\b)/.test(firstLine)) return "python"; - if (/(\bnode\b|\bbun\b|\bdeno\b|\bjavascript\b|\bjs\b)/.test(firstLine)) return "js"; - } - const jsMarkers = - /(^|\n)\s*(const|let|var|async\s+function|function\s*\*?\s*[\w$]*\s*\(|import\s+[^\n]+\sfrom\s|export\s+(default|const|let|function|class|async)|require\s*\(|console\.\w+\s*\(|=>|;\s*$)/m; - const pyMarkers = - /(^|\n)\s*(def\s+\w+\s*\(|from\s+[\w.]+\s+import|import\s+\w+(\s+as\s+\w+)?\s*$|class\s+\w+\s*[(:]|print\s*\(|elif\s+[^\n]*:|with\s+[^\n]+:\s*$|@[\w.]+\s*$)/m; - const hasJs = jsMarkers.test(code); - const hasPy = pyMarkers.test(code); - if (hasJs && !hasPy) return "js"; - if (hasPy && !hasJs) return "python"; - return undefined; -} - async function resolveBackend( session: ToolSession, requested: EvalLanguage | undefined, @@ -180,7 +153,7 @@ async function resolveBackend( return { backend: jsBackend, fallback: false }; } // Auto-detect. - const sniffed = sniffLanguage(code); + const sniffed = sniffEvalLanguage(code); if (sniffed === "python" && allowPy && (await pythonBackend.isAvailable(session))) { return { backend: pythonBackend, fallback: false }; } @@ -446,10 +419,11 @@ export class EvalTool implements AgentTool { pushUpdate(); const errorMsg = result.output || "Command aborted"; const combinedOutput = cellOutputs.join("\n\n"); + const abortSuffix = parsedInput.aborted ? `\n\n${ABORT_WARNING}` : ""; const outputText = - cells.length > 1 + (cells.length > 1 ? `${combinedOutput}\n\nCell ${i + 1} aborted: ${errorMsg}` - : combinedOutput || errorMsg; + : combinedOutput || errorMsg) + abortSuffix; const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput); const details: EvalToolDetails = { @@ -473,12 +447,13 @@ export class EvalTool implements AgentTool { cellResult.status = "error"; pushUpdate(); const combinedOutput = cellOutputs.join("\n\n"); + const abortSuffix = parsedInput.aborted ? `\n\n${ABORT_WARNING}` : ""; const outputText = - cells.length > 1 + (cells.length > 1 ? `${combinedOutput}\n\nCell ${i + 1} failed (exit code ${result.exitCode}). Earlier cells succeeded—their state persists. Fix only cell ${i + 1}.` : combinedOutput ? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}` - : `Command exited with code ${result.exitCode}`; + : `Command exited with code ${result.exitCode}`) + abortSuffix; const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput); const details: EvalToolDetails = { @@ -503,8 +478,10 @@ export class EvalTool implements AgentTool { } const combinedOutput = cellOutputs.join("\n\n"); + const abortSuffix = parsedInput.aborted ? `\n\n${ABORT_WARNING}` : ""; const outputText = - combinedOutput || (jsonOutputs.length > 0 || images.length > 0 ? "(no text output)" : "(no output)"); + (combinedOutput || (jsonOutputs.length > 0 || images.length > 0 ? "(no text output)" : "(no output)")) + + abortSuffix; const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput); const details: EvalToolDetails = { diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index da025e2eb..0d0663bf0 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -18,6 +18,7 @@ import { HL_EDIT_SEP, hashlineEditParamsSchema, parseHashline, + parseHashlineWithWarnings, splitHashlineInput, splitHashlineInputs, tryRecoverHashlineWithCache, @@ -715,3 +716,47 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(cache.get("/tmp/file-31.ts")).not.toBeNull(); }); }); + +describe("hashline *** Abort recovery sentinel (harmony-leak mitigation)", () => { + const sentinel = "*** Abort"; + + it("parser breaks at *** Abort and surfaces a warning", () => { + const diff = [`+ ${tag(1, "alpha")}`, pl("HELLO"), sentinel, `+ ${tag(99, "junk")}`, pl("never")].join("\n"); + const { edits, warnings } = parseHashlineWithWarnings(diff); + expect(edits).toHaveLength(1); + expect(edits[0]).toMatchObject({ kind: "insert", text: "HELLO" }); + expect(warnings.length).toBeGreaterThan(0); + expect(warnings[0]).toMatch(/truncated mid-call/i); + }); + + it("appended sentinel from harmony-leak truncation: ops above are preserved", () => { + // Mirrors the exact shape harmony-leak emits inside a single section. + const diff = `+ ${tag(1, "alpha")}\n${pl("KEPT")}\n*** Abort\n`; + const { edits, warnings } = parseHashlineWithWarnings(diff); + expect(edits).toHaveLength(1); + expect(edits[0]).toMatchObject({ text: "KEPT" }); + expect(warnings.length).toBeGreaterThan(0); + }); + + it("splitter respects *** Abort like *** End Patch", () => { + const input = [ + `@a.ts`, + `+ ${tag(1, "alpha")}`, + pl("a-payload"), + sentinel, + `@b.ts`, + `+ ${tag(1, "beta")}`, + pl("never-emitted"), + ].join("\n"); + const sections = splitHashlineInputs(input); + expect(sections).toHaveLength(1); + expect(sections[0].path).toBe("a.ts"); + expect(sections[0].diff.includes("never-emitted")).toBe(false); + }); + + it("clean input without sentinel produces no warning", () => { + const diff = `+ ${tag(1, "alpha")}\n${pl("PAYLOAD")}\n`; + const { warnings } = parseHashlineWithWarnings(diff); + expect(warnings).toEqual([]); + }); +}); diff --git a/packages/coding-agent/test/core/python-prelude.test.ts b/packages/coding-agent/test/core/python-prelude.test.ts index 81c528197..39b4757e8 100644 --- a/packages/coding-agent/test/core/python-prelude.test.ts +++ b/packages/coding-agent/test/core/python-prelude.test.ts @@ -63,7 +63,7 @@ describe.skipIf(!shouldRun)("PYTHON_PRELUDE integration", () => { ].join("\n"); const result = await tool.execute("tool-call-1", { - input: `===== py:"prelude helpers" =====\n${code}\n`, + input: `*** Begin PY\n*** Title: prelude helpers\n${code}\n*** End PY\n`, }); const output = result.content.find(item => item.type === "text")?.text ?? ""; expect(output).toContain("HELPERS_OK=1"); diff --git a/packages/coding-agent/test/eval/parse.test.ts b/packages/coding-agent/test/eval/parse.test.ts index 1c5750c47..48f77a7fb 100644 --- a/packages/coding-agent/test/eval/parse.test.ts +++ b/packages/coding-agent/test/eval/parse.test.ts @@ -2,228 +2,272 @@ import { describe, expect, it } from "bun:test"; import { parseEvalInput } from "../../src/eval/parse"; describe("parseEvalInput", () => { - it("parses a single header cell with title shorthand and t timeout", () => { - const result = parseEvalInput(`===== py:"setup" t:15s ===== + it("parses a single cell with title and timeout", () => { + const result = parseEvalInput(`*** Begin PY +*** Title: setup +*** Timeout: 15s print("hi") -`); - - expect(result.cells).toEqual([ - { - index: 0, - title: "setup", - code: 'print("hi")', - language: "python", - languageOrigin: "header", - timeoutMs: 15_000, - reset: false, - }, - ]); - }); - - it("treats bare rst as a per-language kernel wipe for that cell", () => { - const result = parseEvalInput(`===== py rst id:"bootstrap" ===== -import json -===== js rst ===== -const x = 1; -`); - - expect(result.cells.map(cell => [cell.language, cell.reset, cell.title])).toEqual([ - ["python", true, "bootstrap"], - ["js", true, undefined], - ]); - }); - - it("inherits language across consecutive cells when omitted", () => { - const result = parseEvalInput(`===== js ===== -const a = 1; -===== ===== -const b = a + 1; -`); - - expect(result.cells.map(cell => [cell.language, cell.languageOrigin, cell.code, cell.reset])).toEqual([ - ["js", "header", "const a = 1;", false], - ["js", "header", "const b = a + 1;", false], - ]); - }); - - it("accepts asymmetric bar runs and case-insensitive language tokens", () => { - const result = parseEvalInput(`===== TypeScript ====== -const a = 1; -====== IPython ===== -print("ipy") -`); - - expect(result.cells.map(cell => [cell.language, cell.languageOrigin])).toEqual([ - ["js", "header"], - ["python", "header"], - ]); - }); - - it("uses canonical id and t attributes, with explicit attrs winning over positional", () => { - const result = parseEvalInput(`===== py 5s some words t:2m id:"explicit win" ===== -print(1) -`); - - expect(result.cells[0]).toMatchObject({ - title: "explicit win", - timeoutMs: 120_000, - language: "python", - }); - }); - - it("accepts fallback aliases for id, t, and rst keys", () => { - const idAliases = ["title", "name", "cell", "file", "label"]; - for (const key of idAliases) { - const result = parseEvalInput(`===== py ${key}:"alpha" =====\nprint(1)\n`); - expect(result.cells[0].title).toBe("alpha"); - } - - const timeoutAliases = ["timeout", "duration", "time"]; - for (const key of timeoutAliases) { - const result = parseEvalInput(`===== py ${key}:2m =====\nprint(1)\n`); - expect(result.cells[0].timeoutMs).toBe(120_000); - } - - const result = parseEvalInput(`===== py reset:true =====\nprint(1)\n`); - expect(result.cells[0].reset).toBe(true); - }); - - it("first occurrence wins when canonical and alias collide", () => { - const canonicalFirst = parseEvalInput(`===== py id:"canon" title:"alias" ===== -print(1) -`); - const aliasFirst = parseEvalInput(`===== py title:"alias" id:"canon" ===== -print(1) -`); - - expect(canonicalFirst.cells[0].title).toBe("canon"); - expect(aliasFirst.cells[0].title).toBe("alias"); - }); - - it("parses millisecond, second, and minute durations", () => { - const result = parseEvalInput(`===== py t:500ms ===== -a = 1 -===== py t:5 ===== -a = 2 -===== py t:2m ===== -a = 3 -`); - - expect(result.cells.map(cell => cell.timeoutMs)).toEqual([500, 5_000, 120_000]); - }); - - it("treats unrecognized header tokens as a title and inherits the language", () => { - const result = parseEvalInput(`===== ruby ===== -puts "no" -`); - - expect(result.cells[0]).toMatchObject({ - title: "ruby", - code: 'puts "no"', - language: "python", - languageOrigin: "default", - }); - }); - - it("joins multiple positional title fragments with spaces", () => { - const result = parseEvalInput(`===== py compute totals ===== -print(1) -`); - - expect(result.cells[0].title).toBe("compute totals"); - }); - - it("accepts back-to-back header cells without blank separators", () => { - const result = parseEvalInput(`===== py id:"a" ===== -print("a") -===== py id:"b" ===== -print("b") -`); - - expect(result.cells.map(cell => [cell.title, cell.code])).toEqual([ - ["a", 'print("a")'], - ["b", 'print("b")'], - ]); - }); - - it("wraps bare code with no headers in a single implicit cell", () => { - const result = parseEvalInput(`print("hello") -print("world") -`); - - expect(result.cells).toEqual([ - { - index: 0, - title: undefined, - code: 'print("hello")\nprint("world")', - language: "python", - languageOrigin: "default", - timeoutMs: 30_000, - reset: false, - }, - ]); - }); - - it("strips blank lines between cells from the preceding cell's code", () => { - const result = parseEvalInput(`===== js ===== -const x = 1; - -===== ===== -const y = 2; -`); - - expect(result.cells.map(cell => [cell.language, cell.languageOrigin, cell.code])).toEqual([ - ["js", "header", "const x = 1;"], - ["js", "header", "const y = 2;"], - ]); - }); - - it("accepts an empty header introducing a default cell with no info", () => { - const result = parseEvalInput(`===== -print("still typing") +*** End PY `); expect(result.cells).toHaveLength(1); expect(result.cells[0]).toMatchObject({ - code: 'print("still typing")', + index: 0, + title: "setup", + code: 'print("hi")', language: "python", - languageOrigin: "default", + languageOrigin: "header", + timeoutMs: 15_000, reset: false, }); }); - it("ignores unknown attribute keys without erroring", () => { - const result = parseEvalInput(`===== py mystery:123 id:"ok" ===== -print(1) + it("treats *** Reset as a per-cell kernel wipe", () => { + const result = parseEvalInput(`*** Begin PY +*** Title: bootstrap +*** Reset +import json +*** End PY +*** Begin JS +*** Reset +const x = 1; +*** End JS `); - expect(result.cells[0]).toMatchObject({ title: "ok", language: "python" }); + expect(result.cells).toHaveLength(2); + expect(result.cells[0]).toMatchObject({ language: "python", title: "bootstrap", reset: true }); + expect(result.cells[1]).toMatchObject({ language: "js", reset: true, title: undefined }); }); - it("rejects an invalid rst value", () => { - expect(() => - parseEvalInput(`===== py rst:maybe ===== -print(1) -`), - ).toThrow("invalid rst value"); - }); - - it("rejects an invalid t value", () => { - expect(() => - parseEvalInput(`===== py t:forever ===== -print(1) -`), - ).toThrow("invalid duration"); - }); - - it("does not treat lines that start with equals but have no closing bar as a header", () => { - const result = parseEvalInput(`===== py ===== -x = 1 -===== not a header -y = 2 + it("accepts JS, TS, and PY language tokens (case-insensitive)", () => { + const result = parseEvalInput(`*** Begin TS +const a = 1; +*** End TS +*** Begin py +print("py") +*** End py `); + expect(result.cells.map(c => c.language)).toEqual(["js", "python"]); + }); + + it("parses millisecond, second, and minute durations", () => { + const result = parseEvalInput(`*** Begin PY +*** Timeout: 500ms +a = 1 +*** End PY +*** Begin PY +*** Timeout: 5 +a = 2 +*** End PY +*** Begin PY +*** Timeout: 2m +a = 3 +*** End PY +`); + + expect(result.cells.map(c => c.timeoutMs)).toEqual([500, 5_000, 120_000]); + }); + + it("attribute order is flexible and only the first wins", () => { + const result = parseEvalInput(`*** Begin PY +*** Timeout: 1s +*** Title: first +*** Title: ignored +*** Timeout: 9s +print(1) +*** End PY +`); + + expect(result.cells[0]).toMatchObject({ title: "first", timeoutMs: 1_000 }); + }); + + it("preserves blank lines inside the cell body", () => { + const result = parseEvalInput(`*** Begin JS +const x = 1; + +const y = 2; +*** End JS +`); + + expect(result.cells[0].code).toBe("const x = 1;\n\nconst y = 2;"); + }); + + it("treats blank lines between cells as separators, not code", () => { + const result = parseEvalInput(`*** Begin PY +print("a") +*** End PY + + +*** Begin PY +print("b") +*** End PY +`); + + expect(result.cells).toHaveLength(2); + expect(result.cells[0].code).toBe('print("a")'); + expect(result.cells[1].code).toBe('print("b")'); + }); + + it("falls back to language sniffing when the begin marker has no recognized language", () => { + const result = parseEvalInput(`*** Begin RUBY +const x = 1; +console.log(x); +*** End +`); + expect(result.cells[0]).toMatchObject({ language: "js", languageOrigin: "default" }); + }); + + it("accepts `**Begin` (two stars) as well as `***Begin`", () => { + const result = parseEvalInput(`**Begin PY +print(1) +**End +`); + expect(result.cells[0]).toMatchObject({ language: "python", code: "print(1)" }); + }); + + it("implicitly closes a cell when a new *** Begin appears without an *** End", () => { + const result = parseEvalInput(`*** Begin PY +print("a") +*** Begin JS +const x = 1; +*** End JS +`); + expect(result.cells).toHaveLength(2); + expect(result.cells[0]).toMatchObject({ language: "python", code: 'print("a")' }); + expect(result.cells[1]).toMatchObject({ language: "js", code: "const x = 1;" }); + }); + + it("ignores the language token on `*** End` (leniency)", () => { + const result = parseEvalInput(`*** Begin PY +print(1) +*** End JS +`); + expect(result.cells[0]).toMatchObject({ language: "python", code: "print(1)" }); + }); + + it("accepts long-form language aliases (Python, JavaScript, TypeScript)", () => { + const result = parseEvalInput(`*** Begin Python +print(1) +*** End +*** begin javascript +const x = 1; +*** End +`); + expect(result.cells.map(c => c.language)).toEqual(["python", "js"]); + }); + + it("tolerates whitespace and case variations on directives", () => { + const result = parseEvalInput(`***\tBegin\tPY +***title: tabby +***\tTimeout:\t250ms +***reset +print(1) +***End +`); + expect(result.cells[0]).toMatchObject({ + title: "tabby", + timeoutMs: 250, + reset: true, + language: "python", + code: "print(1)", + }); + }); + + it("implicitly closes the final cell at EOF when *** End is missing", () => { + const result = parseEvalInput(`*** Begin PY +print(1) +`); expect(result.cells).toHaveLength(1); - expect(result.cells[0].code).toBe("x = 1\n===== not a header\ny = 2"); + expect(result.cells[0]).toMatchObject({ language: "python", code: "print(1)" }); + }); + + it("treats bare code without any *** Begin as a single implicit cell", () => { + const result = parseEvalInput(`def greet():\n print('hi')\ngreet()\n`); + expect(result.cells).toHaveLength(1); + expect(result.cells[0]).toMatchObject({ + language: "python", + languageOrigin: "default", + code: "def greet():\n print('hi')\ngreet()", + }); + }); + + it("strips a markdown code fence wrapper and uses its language tag", () => { + const result = parseEvalInput("```js\nconst x = 1;\n```\n"); + expect(result.cells).toHaveLength(1); + expect(result.cells[0]).toMatchObject({ + language: "js", + languageOrigin: "header", + code: "const x = 1;", + }); + }); + + it("rejects invalid duration", () => { + expect(() => + parseEvalInput(`*** Begin PY +*** Timeout: forever +print(1) +*** End PY +`), + ).toThrow(/invalid duration/); + }); + describe("*** Abort recovery sentinel (harmony-leak mitigation)", () => { + it("drops the in-progress cell and stops parsing", () => { + const result = parseEvalInput(`*** Begin PY +print("a") +*** End PY +*** Begin JS +const partial = 1; /* contamination starts mid-cell */ +*** Abort +*** Begin TS +const never_runs = 1; +`); + expect(result.aborted).toBe(true); + expect(result.cells).toHaveLength(1); + expect(result.cells[0].language).toBe("python"); + expect(result.cells[0].code).toBe('print("a")'); + }); + + it("between cells: keeps preceding cells, sets aborted, drops trailing cells", () => { + const result = parseEvalInput(`*** Begin PY +print("a") +*** End PY + +*** Abort + +*** Begin PY +print("never") +*** End PY +`); + expect(result.aborted).toBe(true); + expect(result.cells).toHaveLength(1); + expect(result.cells[0].code).toBe('print("a")'); + }); + + it("implicit-cell input containing *** Abort is rejected entirely", () => { + const result = parseEvalInput(`print("partial") +*** Abort +`); + expect(result.aborted).toBe(true); + expect(result.cells).toHaveLength(0); + }); + + it("appended sentinel from harmony-leak truncation: abort flag set, prior cell preserved", () => { + // Mirrors the exact shape harmony-leak emits: original input truncated + // at the contaminated line, then "\n*** Abort\n" appended. + const truncated = `*** Begin PY\nprint("ok")\n*** End PY\n*** Abort\n`; + const result = parseEvalInput(truncated); + expect(result.aborted).toBe(true); + expect(result.cells).toHaveLength(1); + expect(result.cells[0].code).toBe('print("ok")'); + }); + + it("absent sentinel: aborted is undefined (not falsely set)", () => { + const result = parseEvalInput(`*** Begin PY +print(1) +*** End PY +`); + expect(result.aborted).toBeUndefined(); + }); }); }); diff --git a/scripts/session-stats/harmony_backtest.py b/scripts/session-stats/harmony_backtest.py new file mode 100755 index 000000000..9b54bff9d --- /dev/null +++ b/scripts/session-stats/harmony_backtest.py @@ -0,0 +1,991 @@ +#!/usr/bin/env python3 +""" +Backtest GPT-5 Harmony-header leak handling against ~/.omp/stats.db. + +This is a dry-run analysis tool. It does not mutate stats.db or session JSONL. +It scans stored assistant/tool-call surfaces, applies a selected detection and +recovery strategy, and prints which edit inputs would be preserved by a +sanitize-tail strategy versus aborted/replayed. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import sqlite3 +import sys +from collections import Counter, defaultdict +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +DB_PATH = Path.home() / ".omp" / "stats.db" + +MARKER_RE = re.compile(r"\bto=functions\.[A-Za-z_][A-Za-z0-9_]*") +HARMONY_RE = re.compile(r"<\|(start|end|channel|message|call|return)\|>") +CHANNEL_WORD_RE = re.compile(r"\b(analysis|commentary|assistant|user|system|developer|tool)\s+to=functions\.") +GLITCH_RE = re.compile(r"\b(changedFiles|RTLU|Jsii(?:_commentary)?|Japgolly|tRTLUfunctions|Joshi_commentary|Japgolly_commentary|jsii_commentary|Jsii_commentary|Jsii)\b") +NULLISH_RE = re.compile(r"\b(undefined|null)\b") +BODY_CASCADE_RE = re.compile(r"\bto=functions\.[A-Za-z_][A-Za-z0-9_]*\s+code(?:\s|$)") +FAKE_RESULT_RE = re.compile( + r"\bto=functions\.[A-Za-z_][A-Za-z0-9_]*(?s:.{0,80}?)code_output\s*\nCell\s+\d+:" +) +FENCE_RE = re.compile(r"^\s*(```+|~~~+)") + +# Python's stdlib re has no Unicode script properties, so keep the exact +# ranges local and explicit. +SCRIPT_RUN_RE = re.compile( + "[" + "\u3400-\u4DBF" # CJK Extension A + "\u4E00-\u9FFF" # CJK Unified Ideographs + "\uF900-\uFAFF" # CJK Compatibility Ideographs + "\u0400-\u04FF" # Cyrillic + "\u0E00-\u0E7F" # Thai + "\u10A0-\u10FF" # Georgian + "\u0530-\u058F" # Armenian + "\u0C80-\u0CFF" # Kannada + "\u0C00-\u0C7F" # Telugu + "\u0900-\u097F" # Devanagari + "\u0600-\u06FF" # Arabic + "\u0D00-\u0D7F" # Malayalam + "]{2,}" +) + +HEADER_RE = re.compile(r"^(?:@(?P\S.*)|\*\*\* Update File:\s+(?P\S.*))\s*$") +BEGIN_PATCH_RE = re.compile(r"^\*\*\* Begin Patch\s*$") +END_PATCH_RE = re.compile(r"^\*\*\* End Patch\s*$") +INSERT_RE = re.compile(r"^[+<]\s*(?PBOF|EOF|[1-9][0-9]*[A-Za-z]{2})(?:\s*~(?P.*))?\s*$") +RANGE_RE = re.compile(r"(?P[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P[1-9][0-9]*[A-Za-z]{2}))?") +DELETE_RE = re.compile(r"^-\s*(?P[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$") +REPLACE_RE = re.compile(r"^=\s*(?P[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$") + +JSON_DECODER = json.JSONDecoder() + + +@dataclass(frozen=True) +class Signal: + cls: str + start: int + end: int + detail: str + + +@dataclass +class MarkerEvidence: + start: int + end: int + classes: set[str] = field(default_factory=lambda: {"M"}) + + @property + def label(self) -> str: + return "+".join(sorted(self.classes)) + + +@dataclass +class EditSection: + target_file: str + op_count: int = 0 + payload_lines: int = 0 + deleted_lines: int = 0 + + +@dataclass +class EditBoundary: + ok: bool + parsed_end: int + reason: str + sections: list[EditSection] = field(default_factory=list) + line_no: int = 0 + + @property + def op_count(self) -> int: + return sum(s.op_count for s in self.sections) + + @property + def payload_lines(self) -> int: + return sum(s.payload_lines for s in self.sections) + + @property + def deleted_lines(self) -> int: + return sum(s.deleted_lines for s in self.sections) + + +@dataclass +class ToolBacktest: + surface: str + row_id: str + session_file: str + seq: int + entry_id: str | None + call_id: str | None + tool_name: str + model: str | None + provider: str | None + action: str + signals: list[str] + signal_offsets: list[int] + text_len: int + parsed_end: int | None = None + removed_len: int = 0 + removed_sha16: str | None = None + removed_preview: str = "" + clean_preview: str = "" + context_preview: str = "" + edit_files: list[str] = field(default_factory=list) + edit_ops: int = 0 + edit_payload_lines: int = 0 + edit_deleted_lines: int = 0 + parse_reason: str = "" + + +@dataclass +class TextBacktest: + surface: str + row_id: str + session_file: str + seq: int + entry_id: str | None + model: str | None + provider: str | None + action: str + signals: list[str] + signal_offsets: list[int] + text_len: int + context_preview: str + + +def open_ro(path: Path) -> sqlite3.Connection: + if not path.exists(): + sys.exit(f"db not found: {path}. Run scripts/session-stats/sync.py first.") + conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True) + conn.row_factory = sqlite3.Row + return conn + + +def commas(n: int) -> str: + return f"{n:,}" + + + +def one_line(text: str, limit: int = 180) -> str: + text = text.replace("\r", "\\r").replace("\n", " | ").replace("\t", "\\t") + if len(text) <= limit: + return text + return text[: max(0, limit - 3)] + "..." + + +def snippet(text: str, pos: int, radius: int = 120) -> str: + lo = max(0, pos - radius) + hi = min(len(text), pos + radius) + prefix = "..." if lo > 0 else "" + suffix = "..." if hi < len(text) else "" + return one_line(prefix + text[lo:hi] + suffix, radius * 2 + 20) + + +def sha16(text: str) -> str: + return hashlib.sha256(text.encode("utf-8", errors="replace")).hexdigest()[:16] + + +def is_inside_fenced_block(text: str, pos: int) -> bool: + """Best-effort Markdown fence context used to avoid doc/test false positives.""" + in_fence = False + for line in text[:pos].splitlines(): + if FENCE_RE.match(line): + in_fence = not in_fence + return in_fence + + +def ascii_ratio(text: str) -> float: + if not text: + return 1.0 + ascii_count = sum(1 for ch in text if ord(ch) < 128) + return ascii_count / len(text) + + +def script_mismatch_near(text: str, start: int, end: int) -> bool: + near = text[max(0, start - 32) : min(len(text), end + 32)] + if not SCRIPT_RUN_RE.search(near): + return False + surrounding = text[max(0, start - 200) : min(len(text), end + 200)] + return ascii_ratio(surrounding) >= 0.85 + + +def marker_evidence_for( + text: str, + marker: re.Match[str], + parsed_end: int | None, + respect_fences: bool, + include_nullish: bool, +) -> MarkerEvidence | None: + start, end = marker.span() + if respect_fences and is_inside_fenced_block(text, start): + return None + + ev = MarkerEvidence(start=start, end=end) + window16 = text[max(0, start - 16) : min(len(text), end + 16)] + window200 = text[start : min(len(text), start + 200)] + + for c in CHANNEL_WORD_RE.finditer(text[max(0, start - 64) : min(len(text), end + 16)]): + absolute_start = max(0, start - 64) + c.start() + absolute_end = max(0, start - 64) + c.end() + if absolute_start <= start < absolute_end: + ev.classes.add("C") + break + + if GLITCH_RE.search(window16): + ev.classes.add("G") + if include_nullish and NULLISH_RE.search(window16): + ev.classes.add("N") + if script_mismatch_near(text, start, end): + ev.classes.add("S") + if BODY_CASCADE_RE.match(window200) and MARKER_RE.search(window200[marker.end() - start :]): + ev.classes.add("B") + if FAKE_RESULT_RE.match(text, start): + ev.classes.add("R") + if parsed_end is not None and start >= parsed_end: + ev.classes.add("T") + return ev + + +def detect_signals( + text: str, + strategy: str, + parsed_end: int | None = None, + respect_fences: bool = True, + include_nullish: bool = False, +) -> tuple[list[Signal], list[MarkerEvidence]]: + signals: list[Signal] = [] + marker_evidence: list[MarkerEvidence] = [] + + for h in HARMONY_RE.finditer(text): + if respect_fences and is_inside_fenced_block(text, h.start()): + continue + signals.append(Signal("H", h.start(), h.end(), h.group(0))) + + for marker in MARKER_RE.finditer(text): + ev = marker_evidence_for(text, marker, parsed_end, respect_fences, include_nullish) + if ev is None: + continue + marker_evidence.append(ev) + if strategy == "marker": + signals.append(Signal(ev.label, ev.start, ev.end, text[ev.start : ev.end])) + elif len(ev.classes) > 1: + signals.append(Signal(ev.label, ev.start, ev.end, text[ev.start : ev.end])) + + if strategy == "tail": + signals = [s for s in signals if s.cls == "H" or "T" in s.cls.split("+")] + + signals.sort(key=lambda s: (s.start, s.end, s.cls)) + marker_evidence.sort(key=lambda e: (e.start, e.end)) + return signals, marker_evidence + + +def complete_json_end(raw: str) -> tuple[bool, int, str]: + try: + _, end = JSON_DECODER.raw_decode(raw) + return True, end, "json-prefix-ok" + except json.JSONDecodeError as exc: + return False, exc.pos, exc.msg + + +def line_spans(text: str) -> list[tuple[str, int, int]]: + out: list[tuple[str, int, int]] = [] + pos = 0 + for raw in text.splitlines(keepends=True): + start = pos + pos += len(raw) + out.append((raw.rstrip("\r\n"), start, pos)) + if text and (text.endswith("\n") or text.endswith("\r")): + return out + if not text: + return [] + # splitlines(keepends=True) already includes the final unterminated line. + return out + +def parse_legacy_diff_boundary(text: str, *, loose_tail: bool = False) -> EditBoundary: + """Best-effort parser for pre-hashline edit inputs. + + Older sessions used `---path` followed by compact hash/range operations + whose replacement payload was raw text, not `~`-prefixed. That format is + not safe enough for production recovery, but this backtest needs to answer + where a tail-only cleaner would cut if we supported those historical rows. + """ + sections: list[EditSection] = [] + cur: EditSection | None = None + parsed_end = 0 + line_no = 0 + in_payload = False + + old_replace = re.compile( + r"^\s*(?P[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)(?P[=+<])(?P.*)$" + ) + old_delete = re.compile( + r"^\s*-(?P[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$" + ) + + for line, _start, end in line_spans(text): + line_no += 1 + if loose_tail and (MARKER_RE.search(line) or HARMONY_RE.search(line)) and cur is not None and cur.op_count > 0: + break + + if line.startswith("---"): + target = line[3:].strip() + if not target: + break + cur = EditSection(target_file=target) + sections.append(cur) + parsed_end = end + in_payload = False + continue + + if cur is None: + if not line.strip(): + parsed_end = end + continue + break + + if line.startswith("+++") or line.startswith("@@"): + if line.startswith("@@"): + cur.op_count += 1 + in_payload = True + else: + in_payload = False + parsed_end = end + continue + + dele = old_delete.match(line) + if dele: + cur.op_count += 1 + cur.deleted_lines += range_deleted_lines(dele.group("range")) + parsed_end = end + in_payload = False + continue + + repl = old_replace.match(line) + if repl: + cur.op_count += 1 + if repl.group("op") == "=": + cur.deleted_lines += range_deleted_lines(repl.group("range")) + if repl.group("tail"): + cur.payload_lines += 1 + parsed_end = end + in_payload = True + continue + + if line.startswith("+") and not line.startswith("+++"): + cur.payload_lines += 1 + parsed_end = end + in_payload = True + continue + + if in_payload and (line.startswith("-") or line.startswith(" ") or line.startswith("\\")): + if line.startswith("-") and not line.startswith("---"): + cur.deleted_lines += 1 + parsed_end = end + continue + + if in_payload: + cur.payload_lines += 1 + parsed_end = end + continue + + if not line.strip(): + parsed_end = end + continue + + break + + return EditBoundary( + ok=parsed_end > 0 and bool(sections), + parsed_end=parsed_end, + reason="legacy-edit-ok" if parsed_end > 0 and sections else "no-complete-edit-prefix", + sections=sections, + line_no=line_no, + ) + + +def anchor_line_no(anchor: str) -> int | None: + m = re.match(r"([1-9][0-9]*)", anchor) + return int(m.group(1)) if m else None + + +def range_deleted_lines(raw_range: str) -> int: + m = RANGE_RE.fullmatch(raw_range) + if not m: + return 1 + a = anchor_line_no(m.group("a")) or 0 + b = anchor_line_no(m.group("b") or m.group("a")) or a + return max(1, b - a + 1) + + +def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBoundary: + sections: list[EditSection] = [] + cur: EditSection | None = None + parsed_end = 0 + line_no = 0 + needs_payload = False + payload_allowed = False + saw_required_payload = False + seen_content = False + + for line, _start, end in line_spans(text): + line_no += 1 + stripped = line.strip() + + if not seen_content and not stripped: + parsed_end = end + continue + seen_content = True + + if BEGIN_PATCH_RE.match(line): + if needs_payload and not saw_required_payload: + break + parsed_end = end + payload_allowed = False + continue + + if END_PATCH_RE.match(line): + if needs_payload and not saw_required_payload: + break + parsed_end = end + payload_allowed = False + continue + + header = HEADER_RE.match(line) + if header: + if needs_payload and not saw_required_payload: + break + target = (header.group("at") or header.group("upd") or "").strip() + cur = EditSection(target_file=target) + sections.append(cur) + parsed_end = end + needs_payload = False + payload_allowed = False + saw_required_payload = False + continue + + if cur is None: + break + + if line.startswith("~"): + if not payload_allowed: + break + cur.payload_lines += 1 + parsed_end = end + saw_required_payload = True + needs_payload = False + continue + + if not stripped: + if needs_payload and not saw_required_payload: + break + parsed_end = end + payload_allowed = False + needs_payload = False + saw_required_payload = False + continue + + if needs_payload and not saw_required_payload: + break + + trimmed = line.lstrip() + ins = INSERT_RE.match(trimmed) + if ins: + cur.op_count += 1 + inline = ins.group("inline") + if inline is None: + needs_payload = True + payload_allowed = True + saw_required_payload = False + # Not complete until at least one payload line appears. + else: + cur.payload_lines += 1 + parsed_end = end + needs_payload = False + payload_allowed = False + saw_required_payload = False + continue + + dele = DELETE_RE.match(trimmed) + if dele: + cur.op_count += 1 + cur.deleted_lines += range_deleted_lines(dele.group("range")) + parsed_end = end + needs_payload = False + payload_allowed = False + saw_required_payload = False + continue + + repl = REPLACE_RE.match(trimmed) + if repl: + cur.op_count += 1 + cur.deleted_lines += range_deleted_lines(repl.group("range")) + parsed_end = end + needs_payload = False + payload_allowed = True + saw_required_payload = False + continue + + break + + reason = "ok" if parsed_end > 0 and sections else "no-complete-edit-prefix" + if needs_payload and not saw_required_payload: + reason = "insert-missing-payload" + if not sections and text.lstrip().startswith("---"): + return parse_legacy_diff_boundary(text, loose_tail=legacy_loose_tail) + return EditBoundary( + ok=parsed_end > 0 and bool(sections), + parsed_end=parsed_end, + reason=reason, + sections=sections, + line_no=line_no, + ) + + +def parse_arg_json(raw: str) -> tuple[Any | None, bool, str]: + try: + return json.loads(raw), True, "ok" + except json.JSONDecodeError as exc: + return None, False, f"json-error:{exc.pos}:{exc.msg}" + + +def extract_primary_text(tool_name: str, arg_json: str, parsed: Any | None) -> tuple[str, str]: + if tool_name == "edit" and isinstance(parsed, dict) and isinstance(parsed.get("input"), str): + return "edit.input", parsed["input"] + if tool_name == "eval" and isinstance(parsed, dict) and isinstance(parsed.get("input"), str): + return "eval.input", parsed["input"] + if tool_name == "write" and isinstance(parsed, dict) and isinstance(parsed.get("content"), str): + return "write.content", parsed["content"] + if tool_name == "bash" and isinstance(parsed, dict) and isinstance(parsed.get("command"), str): + return "bash.command", parsed["command"] + return "arg_json", arg_json + + + +def action_for_tool( + tool_name: str, + surface: str, + text: str, + signals: list[Signal], + boundary: EditBoundary | None, +) -> str: + if not signals: + return "allow" + if tool_name == "edit" and surface == "edit.input" and boundary is not None and boundary.ok: + if all(sig.start >= boundary.parsed_end for sig in signals): + return "sanitize_tail" + return "abort_replay" + return "abort_replay" + + +def evaluate_tool_row( + row: sqlite3.Row, + strategy: str, + respect_fences: bool, + include_nullish: bool, + legacy_loose_tail: bool = False, +) -> ToolBacktest: + arg_json = row["arg_json"] or "" + tool_name = row["tool_name"] or "" + parsed, json_ok, json_reason = parse_arg_json(arg_json) + surface, text = extract_primary_text(tool_name, arg_json, parsed) + + boundary: EditBoundary | None = None + parsed_end: int | None = None + parse_reason = json_reason + if tool_name == "edit" and surface == "edit.input": + boundary = parse_edit_boundary(text, legacy_loose_tail=legacy_loose_tail) + parsed_end = boundary.parsed_end if boundary.ok else None + parse_reason = boundary.reason + else: + ok, end, reason = complete_json_end(arg_json) + if ok: + parsed_end = end + parse_reason = reason if json_ok else json_reason + + signals, _marker_evidence = detect_signals( + text, + strategy=strategy, + parsed_end=parsed_end, + respect_fences=respect_fences, + include_nullish=include_nullish, + ) + action = action_for_tool(tool_name, surface, text, signals, boundary) + + removed = "" + clean_preview = "" + edit_files: list[str] = [] + edit_ops = 0 + edit_payload_lines = 0 + edit_deleted_lines = 0 + + if boundary is not None: + edit_files = [s.target_file for s in boundary.sections] + edit_ops = boundary.op_count + edit_payload_lines = boundary.payload_lines + edit_deleted_lines = boundary.deleted_lines + if action == "sanitize_tail": + removed = text[boundary.parsed_end :] + cleaned = text[: boundary.parsed_end] + clean_preview = tail_preview(cleaned) + + first_pos = signals[0].start if signals else 0 + return ToolBacktest( + surface=surface, + row_id=str(row["id"]), + session_file=row["session_file"], + seq=row["seq"], + entry_id=row["entry_id"], + call_id=row["call_id"], + tool_name=tool_name, + model=row["model"], + provider=row["provider"], + action=action, + signals=[s.cls for s in signals], + signal_offsets=[s.start for s in signals], + text_len=len(text), + parsed_end=parsed_end, + removed_len=len(removed), + removed_sha16=sha16(removed) if removed else None, + removed_preview=one_line(removed, 200) if removed else "", + clean_preview=clean_preview, + context_preview=snippet(text, first_pos) if signals else "", + edit_files=edit_files, + edit_ops=edit_ops, + edit_payload_lines=edit_payload_lines, + edit_deleted_lines=edit_deleted_lines, + parse_reason=parse_reason, + ) + + +def tail_preview(text: str, max_lines: int = 8, limit: int = 420) -> str: + lines = text.splitlines() + tail = "\n".join(lines[-max_lines:]) + return one_line(tail, limit) + + +def evaluate_text_row( + row: sqlite3.Row, + surface: str, + text: str, + strategy: str, + respect_fences: bool, + include_nullish: bool, +) -> TextBacktest: + signals, _ = detect_signals( + text, + strategy=strategy, + parsed_end=None, + respect_fences=respect_fences, + include_nullish=include_nullish, + ) + action = "rewrite_candidate" if signals else "allow" + first_pos = signals[0].start if signals else 0 + return TextBacktest( + surface=surface, + row_id=f"{row['session_file']}:{row['seq']}:{surface}", + session_file=row["session_file"], + seq=row["seq"], + entry_id=row["entry_id"], + model=row["model"], + provider=row["provider"], + action=action, + signals=[s.cls for s in signals], + signal_offsets=[s.start for s in signals], + text_len=len(text), + context_preview=snippet(text, first_pos) if signals else "", + ) + + +def candidate_where(column: str) -> str: + return " OR ".join( + [ + f"{column} LIKE '%to=functions.%'", + f"{column} LIKE '%<|start|>%'", + f"{column} LIKE '%<|end|>%'", + f"{column} LIKE '%<|channel|>%'", + f"{column} LIKE '%<|message|>%'", + f"{column} LIKE '%<|call|>%'", + f"{column} LIKE '%<|return|>%'", + ] + ) + + +def scan_tools(conn: sqlite3.Connection, args: argparse.Namespace) -> list[ToolBacktest]: + where = candidate_where("arg_json") + params: list[Any] = [] + if args.provider: + where = f"({where}) AND provider = ?" + params.append(args.provider) + if args.model: + where = f"({where}) AND model = ?" + params.append(args.model) + if args.tool: + where = f"({where}) AND tool_name = ?" + params.append(args.tool) + sql = f""" + SELECT id, session_file, seq, entry_id, call_id, tool_name, raw_tool_name, + timestamp, model, provider, arg_json + FROM ss_tool_calls + WHERE {where} + ORDER BY timestamp, id + """ + rows = conn.execute(sql, params).fetchall() + return [ + evaluate_tool_row( + row, + strategy=args.strategy, + respect_fences=not args.no_fence_context, + include_nullish=args.include_nullish, + legacy_loose_tail=args.legacy_loose_tail, + ) + for row in rows + ] + + +def scan_assistant(conn: sqlite3.Connection, args: argparse.Namespace) -> list[TextBacktest]: + if not args.include_assistant: + return [] + text_where = candidate_where("text_blob") + thinking_where = candidate_where("thinking_blob") + where = f"({text_where}) OR ({thinking_where})" + params: list[Any] = [] + if args.provider: + where = f"({where}) AND provider = ?" + params.append(args.provider) + if args.model: + where = f"({where}) AND model = ?" + params.append(args.model) + sql = f""" + SELECT session_file, seq, entry_id, timestamp, model, provider, + text_blob, thinking_blob + FROM ss_assistant_msgs + WHERE {where} + ORDER BY timestamp, session_file, seq + """ + out: list[TextBacktest] = [] + for row in conn.execute(sql, params): + if row["text_blob"]: + out.append( + evaluate_text_row( + row, + "assistant_text", + row["text_blob"], + args.strategy, + not args.no_fence_context, + args.include_nullish, + ) + ) + if row["thinking_blob"]: + out.append( + evaluate_text_row( + row, + "assistant_thinking", + row["thinking_blob"], + args.strategy, + not args.no_fence_context, + args.include_nullish, + ) + ) + return out + + +def print_counter(title: str, counter: Counter[str]) -> None: + print(title) + for key, value in counter.most_common(): + print(f" {key:<32} {value:>6}") + + +def print_tool_summary(results: list[ToolBacktest]) -> None: + print("=== tool-call scan ===") + print(f"candidate rows: {commas(len(results))}") + print_counter("\nby action:", Counter(r.action for r in results)) + print_counter("\nby tool/action:", Counter(f"{r.tool_name}:{r.action}" for r in results)) + print_counter("\nby model/action:", Counter(f"{r.model or ''}:{r.action}" for r in results)) + signal_counter: Counter[str] = Counter() + for r in results: + if r.signals: + signal_counter.update(r.signals) + else: + signal_counter["none"] += 1 + print_counter("\nby signal:", signal_counter) + + edit_results = [r for r in results if r.tool_name == "edit"] + if edit_results: + sanitized = sum(1 for r in edit_results if r.action == "sanitize_tail") + aborted = sum(1 for r in edit_results if r.action == "abort_replay") + print("\nedit preservation:") + print(f" edit candidates: {commas(len(edit_results))}") + print(f" sanitize_tail: {commas(sanitized)}") + print(f" abort_replay: {commas(aborted)}") + if sanitized: + preserved_ops = sum(r.edit_ops for r in edit_results if r.action == "sanitize_tail") + preserved_payload = sum(r.edit_payload_lines for r in edit_results if r.action == "sanitize_tail") + removed = sum(r.removed_len for r in edit_results if r.action == "sanitize_tail") + print(f" ops preserved by sanitize: {commas(preserved_ops)}") + print(f" payload lines preserved: {commas(preserved_payload)}") + print(f" tail bytes removed: {commas(removed)}") + + +def print_text_summary(results: list[TextBacktest]) -> None: + if not results: + return + print("\n=== assistant message scan ===") + print(f"candidate surfaces: {commas(len(results))}") + print_counter("\nby action:", Counter(r.action for r in results)) + print_counter("\nby surface/action:", Counter(f"{r.surface}:{r.action}" for r in results)) + print_counter("\nby model/action:", Counter(f"{r.model or ''}:{r.action}" for r in results)) + + + +def signal_summary(labels: list[str], limit: int = 6) -> str: + if not labels: + return "none" + counts = Counter(labels) + parts = [f"{label}x{count}" if count > 1 else label for label, count in counts.most_common(limit)] + rest = sum(counts.values()) - sum(count for _, count in counts.most_common(limit)) + if rest: + parts.append(f"+{rest} more") + return ",".join(parts) + +def print_examples(results: list[ToolBacktest], show: int) -> None: + if show <= 0: + return + print(f"\n=== sanitize_tail edit examples (up to {show}) ===") + sanitize_examples = [r for r in results if r.tool_name == "edit" and r.action == "sanitize_tail"] + for r in sanitize_examples[:show]: + print(f"\n[id={r.row_id} seq={r.seq} model={r.model} signals={signal_summary(r.signals)}]") + print(f"session: {r.session_file}") + print(f"file(s): {', '.join(r.edit_files) if r.edit_files else ''}") + print( + f"parsed_end={r.parsed_end} text_len={r.text_len} removed={r.removed_len} " + f"sha16={r.removed_sha16} ops={r.edit_ops} payload={r.edit_payload_lines}" + ) + print(f"last clean lines: {r.clean_preview}") + print(f"removed preview: {r.removed_preview}") + + print(f"\n=== abort_replay examples (up to {show}) ===") + abort_examples = [r for r in results if r.action == "abort_replay"] + for r in abort_examples[:show]: + print(f"\n[id={r.row_id} tool={r.tool_name} surface={r.surface} seq={r.seq} model={r.model} signals={signal_summary(r.signals)}]") + print(f"session: {r.session_file}") + if r.tool_name == "edit": + print( + f"parse={r.parse_reason} parsed_end={r.parsed_end} text_len={r.text_len} " + f"ops={r.edit_ops} payload={r.edit_payload_lines} files={', '.join(r.edit_files) if r.edit_files else ''}" + ) + print(f"context: {r.context_preview}") + + +def write_json_report(path: Path, tools: list[ToolBacktest], texts: list[TextBacktest]) -> None: + def tool_dict(r: ToolBacktest) -> dict[str, Any]: + return { + "surface": r.surface, + "row_id": r.row_id, + "session_file": r.session_file, + "seq": r.seq, + "entry_id": r.entry_id, + "call_id": r.call_id, + "tool_name": r.tool_name, + "model": r.model, + "provider": r.provider, + "action": r.action, + "signals": r.signals, + "signal_offsets": r.signal_offsets, + "text_len": r.text_len, + "parsed_end": r.parsed_end, + "removed_len": r.removed_len, + "removed_sha16": r.removed_sha16, + "removed_preview": r.removed_preview, + "clean_preview": r.clean_preview, + "context_preview": r.context_preview, + "edit_files": r.edit_files, + "edit_ops": r.edit_ops, + "edit_payload_lines": r.edit_payload_lines, + "edit_deleted_lines": r.edit_deleted_lines, + "parse_reason": r.parse_reason, + } + + def text_dict(r: TextBacktest) -> dict[str, Any]: + return { + "surface": r.surface, + "row_id": r.row_id, + "session_file": r.session_file, + "seq": r.seq, + "entry_id": r.entry_id, + "model": r.model, + "provider": r.provider, + "action": r.action, + "signals": r.signals, + "signal_offsets": r.signal_offsets, + "text_len": r.text_len, + "context_preview": r.context_preview, + } + + payload = { + "tool_calls": [tool_dict(r) for r in tools], + "assistant_surfaces": [text_dict(r) for r in texts], + } + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def main() -> int: + ap = argparse.ArgumentParser( + description="Backtest Harmony leak detection/recovery against session-stats sqlite tables." + ) + ap.add_argument("--db", type=Path, default=DB_PATH, help="stats sqlite path") + ap.add_argument( + "--strategy", + choices=("marker", "fusion", "tail"), + default="fusion", + help=( + "marker = trip on bare marker/control token; " + "fusion = H or M plus co-signal; " + "tail = only H or marker after parsed boundary" + ), + ) + ap.add_argument("--provider", default=None, help="restrict tool/message rows to a provider") + ap.add_argument("--model", default=None, help="restrict tool/message rows to a model") + ap.add_argument("--tool", default=None, help="restrict tool-call rows to one tool") + ap.add_argument("--include-assistant", action="store_true", help="also scan assistant text/thinking surfaces") + ap.add_argument("--include-nullish", action="store_true", help="treat adjacent null/undefined as signal N") + ap.add_argument("--no-fence-context", action="store_true", help="do not exempt Markdown fenced blocks") + ap.add_argument("--legacy-loose-tail", action="store_true", help="model old raw-payload edit inputs as tail-sanitizable at first marker line") + ap.add_argument("--show", type=int, default=8, help="examples per action group") + ap.add_argument("--json-out", type=Path, default=None, help="write machine-readable report") + args = ap.parse_args() + + conn = open_ro(args.db) + print("=== harmony leak backtest ===") + print(f"db: {args.db}") + print(f"strategy: {args.strategy}") + print(f"fences: {'ignored for action' if not args.no_fence_context else 'scanned as active text'}") + if args.legacy_loose_tail: + print("legacy: loose tail mode") + if args.provider: + print(f"provider: {args.provider}") + if args.model: + print(f"model: {args.model}") + if args.tool: + print(f"tool: {args.tool}") + print() + + tool_results = scan_tools(conn, args) + text_results = scan_assistant(conn, args) + + print_tool_summary(tool_results) + print_text_summary(text_results) + print_examples(tool_results, args.show) + + if args.json_out is not None: + write_json_report(args.json_out, tool_results, text_results) + print(f"\nwrote JSON report: {args.json_out}") + + return 0 + + +if __name__ == "__main__": + sys.exit(main())