diff --git a/packages/coding-agent/src/advisor/runtime.ts b/packages/coding-agent/src/advisor/runtime.ts index b48751aa8..0e0c2a0a5 100644 --- a/packages/coding-agent/src/advisor/runtime.ts +++ b/packages/coding-agent/src/advisor/runtime.ts @@ -5,7 +5,8 @@ import * as AIError from "@oh-my-pi/pi-ai/error"; import { raceWithSignal } from "@oh-my-pi/pi-ai/utils/abort"; import { type CursorExecResolvedCarrier, kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; import { logger } from "@oh-my-pi/pi-utils"; -import { obfuscateToolArguments, type SecretObfuscator } from "../secrets/obfuscator"; +import { obfuscateToolArguments } from "../secrets/message-transform"; +import type { SecretObfuscator } from "../secrets/obfuscator"; import { formatExecutionSourcePreview, formatSessionHistoryMarkdown, diff --git a/packages/coding-agent/src/export/share.ts b/packages/coding-agent/src/export/share.ts index 11012e838..1d1ed9e94 100644 --- a/packages/coding-agent/src/export/share.ts +++ b/packages/coding-agent/src/export/share.ts @@ -24,7 +24,8 @@ import type { AssistantMessage, ImageContent, TextContent } from "@oh-my-pi/pi-a import { $which, logger } from "@oh-my-pi/pi-utils"; import { DEFAULT_SHARE_URL } from "@oh-my-pi/pi-wire"; import { $ } from "bun"; -import { obfuscateToolArguments, type SecretObfuscator } from "../secrets/obfuscator"; +import { obfuscateToolArguments } from "../secrets/message-transform"; +import type { SecretObfuscator } from "../secrets/obfuscator"; import { type SessionEntry, type SessionHeader, TITLE_CHANGE_ENTRY_TYPE } from "../session/session-entries"; import type { SessionManager } from "../session/session-manager"; import type { OutputMeta } from "../tools/output-meta"; diff --git a/packages/coding-agent/src/secrets/index.ts b/packages/coding-agent/src/secrets/index.ts index 0c3501528..cb131bf88 100644 --- a/packages/coding-agent/src/secrets/index.ts +++ b/packages/coding-agent/src/secrets/index.ts @@ -4,14 +4,10 @@ import * as path from "node:path"; import { SENSITIVE_TOKEN_RE } from "@oh-my-pi/pi-ai/providers/transform-messages"; import { getSecretPlaceholderKeyPath, isEnoent, logger } from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; -import { - regexHasUnresolvableShortMatchFallback, - type SecretEntry, - SecretObfuscator, - sanitizeSecretFriendlyName, - secretEntriesNeedPlaceholderKey, -} from "./obfuscator"; +import { type SecretEntry, SecretObfuscator } from "./obfuscator"; +import { sanitizeSecretFriendlyName, secretEntriesNeedPlaceholderKey } from "./placeholder"; import { compileSecretRegex } from "./regex"; +import { regexHasUnresolvableShortMatchFallback } from "./replacement"; const PLACEHOLDER_KEY_RE = /^[A-Za-z0-9_-]{43}$/; const cachedPlaceholderKeys = new Map(); @@ -158,11 +154,9 @@ export { deobfuscateToolArguments, obfuscateMessages, obfuscateProviderContext, - type SecretEntry, - SecretObfuscator, - secretEntriesNeedPlaceholderKey, - secretEntryNeedsPlaceholderKey, -} from "./obfuscator"; +} from "./message-transform"; +export { type SecretEntry, SecretObfuscator } from "./obfuscator"; +export { secretEntriesNeedPlaceholderKey, secretEntryNeedsPlaceholderKey } from "./placeholder"; /** * Load secrets from project-local and global secrets.yml files. diff --git a/packages/coding-agent/src/secrets/message-transform.ts b/packages/coding-agent/src/secrets/message-transform.ts new file mode 100644 index 000000000..9ef842d10 --- /dev/null +++ b/packages/coding-agent/src/secrets/message-transform.ts @@ -0,0 +1,287 @@ +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage, Context, ImageContent, Message, TextContent } from "@oh-my-pi/pi-ai"; +import type { SessionContext } from "../session/session-context"; +import type { JsonValue, SecretObfuscator } from "./obfuscator"; +import { collectJsonRegexSecretValues, mapJsonStrings } from "./placeholder-scan"; + +// ═══════════════════════════════════════════════════════════════════════════ +// Display restore (inbound, persisted/provider → local display) +// ═══════════════════════════════════════════════════════════════════════════ + +/** + * Restore secret placeholders for local display. Only message kinds the model + * itself authored from obfuscated context carry placeholders — assistant + * content and the LLM-written branch/compaction summaries. User, developer, and + * tool-result messages are persisted with their literal text, so operator-authored + * placeholder-shaped text must survive untouched; those roles are never walked. + */ +export function deobfuscateSessionContext( + sessionContext: SessionContext, + obfuscator: SecretObfuscator | undefined, +): SessionContext { + if (!obfuscator?.hasSecrets()) return sessionContext; + const messages = deobfuscateAgentMessages(obfuscator, sessionContext.messages); + return messages === sessionContext.messages ? sessionContext : { ...sessionContext, messages }; +} + +export function deobfuscateAgentMessages(obfuscator: SecretObfuscator, messages: AgentMessage[]): AgentMessage[] { + const deob = (text: string): string => obfuscator.deobfuscate(text); + let changed = false; + const result = messages.map((message): AgentMessage => { + switch (message.role) { + case "assistant": { + const content = deobfuscateAssistantContent(obfuscator, message.content); + if (content === message.content) return message; + changed = true; + return { ...message, content }; + } + case "branchSummary": { + const summary = deob(message.summary); + if (summary === message.summary) return message; + changed = true; + return { ...message, summary }; + } + case "compactionSummary": { + const summary = deob(message.summary); + const shortSummary = message.shortSummary === undefined ? undefined : deob(message.shortSummary); + const blocks = message.blocks === undefined ? undefined : deobfuscateTextBlocks(obfuscator, message.blocks); + if (summary === message.summary && shortSummary === message.shortSummary && blocks === message.blocks) { + return message; + } + changed = true; + return { ...message, summary, shortSummary, blocks }; + } + default: + return message; + } + }); + return changed ? result : messages; +} + +/** + * Restore placeholders in assistant content: visible text and tool-call + * arguments/intent/rawBlock. Thinking and signatures are opaque + * provider-replay/hidden-reasoning data and pass through byte-identical. + */ +export function deobfuscateAssistantContent( + obfuscator: SecretObfuscator, + content: AssistantMessage["content"], +): AssistantMessage["content"] { + if (!obfuscator.hasSecrets()) return content; + const deob = (text: string): string => obfuscator.deobfuscate(text); + let changed = false; + const result = content.map((block): AssistantMessage["content"][number] => { + if (block.type === "text") { + const text = deob(block.text); + if (text === block.text) return block; + changed = true; + return { ...block, text }; + } + + if (block.type === "toolCall") { + const args = deobfuscateToolArguments(obfuscator, block.arguments); + const intent = block.intent === undefined ? undefined : deob(block.intent); + const rawBlock = block.rawBlock === undefined ? undefined : deob(block.rawBlock); + if (args === block.arguments && intent === block.intent && rawBlock === block.rawBlock) return block; + changed = true; + return { ...block, arguments: args, intent, rawBlock }; + } + return block; + }); + return changed ? result : content; +} + +/** + * Restore placeholders inside a tool call's arguments. Arguments are arbitrary + * model-authored JSON, so tool-call arguments are the ONLY place a recursive + * JSON walk runs. + */ +export function deobfuscateToolArguments( + obfuscator: SecretObfuscator, + args: Record, +): Record { + if (!obfuscator.hasSecrets()) return args; + return mapJsonStrings(args as JsonValue, s => obfuscator.deobfuscate(s)) as Record; +} + +/** Redact secrets inside a tool call's arguments (same JSON-walk exception as {@link deobfuscateToolArguments}). */ +export function obfuscateToolArguments( + obfuscator: SecretObfuscator, + args: Record, + sharedRegexSecretValues?: ReadonlySet, +): Record { + if (!obfuscator.hasSecrets()) return args; + const regexSecretValues = sharedRegexSecretValues ?? collectJsonRegexSecretValues(obfuscator, args as JsonValue); + return mapJsonStrings(args as JsonValue, s => obfuscator.obfuscate(s, regexSecretValues)) as Record; +} + +// ═══════════════════════════════════════════════════════════════════════════ +// Outbound obfuscation (local → provider) +// ═══════════════════════════════════════════════════════════════════════════ + +type UserFacingMessage = Extract; + +/** Obfuscate `text` blocks of a content array; image and other blocks pass through. */ +function obfuscateTextBlocks( + obfuscator: SecretObfuscator, + content: (TextContent | ImageContent)[], + sharedRegexSecretValues?: ReadonlySet, +): (TextContent | ImageContent)[] { + let changed = false; + const result = content.map((block): TextContent | ImageContent => { + if (block.type !== "text") return block; + const text = obfuscator.obfuscate(block.text, sharedRegexSecretValues); + if (text === block.text) return block; + changed = true; + return { ...block, text }; + }); + return changed ? result : content; +} + +/** Restore placeholders in `text` blocks of a content array; image and other blocks pass through. */ +function deobfuscateTextBlocks( + obfuscator: SecretObfuscator, + content: (TextContent | ImageContent)[], +): (TextContent | ImageContent)[] { + let changed = false; + const result = content.map((block): TextContent | ImageContent => { + if (block.type !== "text") return block; + const text = obfuscator.deobfuscate(block.text); + if (text === block.text) return block; + changed = true; + return { ...block, text }; + }); + return changed ? result : content; +} + +/** + * Re-obfuscate assistant content before it returns to a provider after session + * restoration, removing friendly prefixes made unsafe by this batch. A changed + * thinking block loses its byte-bound replay signature. + */ +function obfuscateAssistantContentForReplay( + obfuscator: SecretObfuscator, + content: AssistantMessage["content"], + sharedRegexSecretValues: ReadonlySet, +): AssistantMessage["content"] { + const obfuscate = (text: string): string => + obfuscator.stripUnsafeFriendlyPlaceholderPrefixes( + obfuscator.obfuscate(text, sharedRegexSecretValues), + sharedRegexSecretValues, + ); + let changed = false; + const result = content.map((block): AssistantMessage["content"][number] => { + if (block.type === "text") { + const text = obfuscate(block.text); + if (text === block.text) return block; + changed = true; + return { ...block, text }; + } + if (block.type === "thinking") { + const thinking = obfuscate(block.thinking); + if (thinking === block.thinking) return block; + changed = true; + return { ...block, thinking, thinkingSignature: undefined }; + } + if (block.type === "toolCall") { + const args = mapJsonStrings(block.arguments as JsonValue, obfuscate) as Record; + const intent = block.intent === undefined ? undefined : obfuscate(block.intent); + const rawBlock = block.rawBlock === undefined ? undefined : obfuscate(block.rawBlock); + if (args === block.arguments && intent === block.intent && rawBlock === block.rawBlock) return block; + changed = true; + return { ...block, arguments: args, intent, rawBlock }; + } + return block; + }); + return changed ? result : content; +} + +function collectMessageRegexSecretValues(obfuscator: SecretObfuscator, messages: Message[]): Set { + const values = new Set(); + const addText = (text: string | undefined): void => { + if (text === undefined) return; + for (const value of obfuscator.collectRegexSecretValuesForObfuscation(text)) { + values.add(value); + } + }; + for (const message of messages) { + if (message.role === "assistant") { + for (const block of message.content) { + if (block.type === "text") addText(block.text); + else if (block.type === "thinking") addText(block.thinking); + else if (block.type === "toolCall") { + for (const value of collectJsonRegexSecretValues(obfuscator, block.arguments as JsonValue)) { + values.add(value); + } + addText(block.intent); + addText(block.rawBlock); + } + } + continue; + } + if ( + message.role !== "user" && + message.role !== "toolResult" && + !(message.role === "developer" && message.attribution === "user") + ) { + continue; + } + const target = message as UserFacingMessage; + if (typeof target.content === "string") { + addText(target.content); + continue; + } + for (const block of target.content) { + if (block.type === "text") addText(block.text); + } + } + return values; +} + +/** + * Redact secrets from outbound messages. User messages, tool results, and + * user-authored developer messages (e.g. `@file` mentions) are obfuscated. + * Assistant replay content is re-obfuscated too, because session restoration + * expands keyed placeholders locally before the next provider request. Inline + * image bytes are never walked. + */ +export function obfuscateMessages(obfuscator: SecretObfuscator, messages: Message[]): Message[] { + if (!obfuscator.hasSecrets()) return messages; + const sharedRegexSecretValues = collectMessageRegexSecretValues(obfuscator, messages); + let changed = false; + const result = messages.map((message): Message => { + if ( + message.role !== "user" && + message.role !== "toolResult" && + !(message.role === "developer" && message.attribution === "user") + ) { + if (message.role !== "assistant") return message; + const content = obfuscateAssistantContentForReplay(obfuscator, message.content, sharedRegexSecretValues); + if (content === message.content) return message; + changed = true; + return { ...message, content }; + } + const target = message as UserFacingMessage; + if (typeof target.content === "string") { + const content = obfuscator.obfuscate(target.content, sharedRegexSecretValues); + if (content === target.content) return message; + changed = true; + return { ...target, content } as Message; + } + const content = obfuscateTextBlocks(obfuscator, target.content, sharedRegexSecretValues); + if (content === target.content) return message; + changed = true; + return { ...target, content } as Message; + }); + return changed ? result : messages; +} + +/** + * Redact outbound provider context. Only conversation messages are rewritten; + * the static system prompt and tool schemas pass through unchanged. + */ +export function obfuscateProviderContext(obfuscator: SecretObfuscator | undefined, context: Context): Context { + if (!obfuscator?.hasSecrets()) return context; + const messages = obfuscateMessages(obfuscator, context.messages); + return messages === context.messages ? context : { ...context, messages }; +} diff --git a/packages/coding-agent/src/secrets/obfuscator.ts b/packages/coding-agent/src/secrets/obfuscator.ts index 5bbda26b8..4bfedc7a0 100644 --- a/packages/coding-agent/src/secrets/obfuscator.ts +++ b/packages/coding-agent/src/secrets/obfuscator.ts @@ -1,8 +1,43 @@ -import * as crypto from "node:crypto"; -import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; -import type { AssistantMessage, Context, ImageContent, Message, TextContent } from "@oh-my-pi/pi-ai"; -import type { SessionContext } from "../session/session-context"; +import { + buildHashBase, + buildKeyedReplacementRun, + buildPlaceholder, + defaultPlaceholderKey, + inferCaseHint, + lookupFriendlyPlaceholderAlias, + MIN_OBFUSCATE_SECRET_LEN, + PLACEHOLDER_RE, + placeholderWithoutFriendlyName, + resumePlaceholderScanAfterRejectedCandidate, + sanitizedLabelCollidesWithSecret, + sanitizeForCollisionCheck, + sanitizeSecretFriendlyName, +} from "./placeholder"; +import { + buildReplaceRegexScan, + countOutsidePlaceholderRanges, + deepWalkStrings, + deobfuscateGeneratedPlaceholderRanges, + extendPastAdjacentPlaceholders, + firstOutsidePlaceholderRange, + mapReplaceRegexMatch, + outsidePlaceholderRangesAnyIndependentlyMatch, + placeholderInnerText, + redactWithFixedReplacementOutsidePlaceholders, + replaceRange, + textOutsidePlaceholderRanges, + trailingOutsidePreservedPlaceholderChunk, + transformOutsidePlaceholdersTracked, +} from "./placeholder-scan"; import { compileSecretRegex } from "./regex"; +import { + ensureDistinctReplacement, + findNonMatchingReplacement, + generateDeterministicReplacement, + type RegexMatchContext, + regexHasUnresolvableShortMatchFallback, + regexRematchesInContext, +} from "./replacement"; // ═══════════════════════════════════════════════════════════════════════════ // Types @@ -20,525 +55,6 @@ export interface SecretEntry { export type JsonValue = string | number | boolean | null | JsonValue[] | { [key: string]: JsonValue | undefined }; export type JsonRecord = { [key: string]: JsonValue | undefined }; -// ═══════════════════════════════════════════════════════════════════════════ -// Deterministic replacement generation -// ═══════════════════════════════════════════════════════════════════════════ - -const REPLACEMENT_CHARS = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789"; -const NONMATCHING_REPLACEMENT_CHARS = `${REPLACEMENT_CHARS}!#$%&()*+,-./:;<=>?@[]^_{|}~`; -// Whitespace bytes used to build last-resort redactions for a default replace -// regex that matches every non-whitespace candidate (e.g. `\S{n}`). Only -// `space`/`tab` are used — never a line terminator — so a `.`-style -// match-everything regex (which matches space and tab but not `\n`) still -// exhausts to the sentinel instead of redacting to a newline run. -const WHITESPACE_REPLACEMENT_CHARS = " \t"; - -/** Generate a deterministic same-length replacement string from a secret value. */ -function generateDeterministicReplacement(secret: string): string { - if (secret.length === 0) return ""; - // Prefix generated chunks with a fixed `ZZ` so re-redacting an already-emitted - // 1–2 char chunk is a fixed point (the deterministic replacement of a <=2-char - // value is itself `Z`/`ZZ`), keeping short default-replacement remainders next - // to a reversible placeholder stable across an obfuscator restart. - const hash = BigInt(Bun.hash(secret)); - const chars = secret.length === 1 ? ["Z"] : ["Z", "Z"]; - let h = hash; - for (let i = chars.length; i < secret.length; i++) { - h = h ^ (BigInt(i + 1) * 0x9e3779b97f4a7c15n); - const idx = Number((h < 0n ? -h : h) % BigInt(REPLACEMENT_CHARS.length)); - chars.push(REPLACEMENT_CHARS[idx]); - } - return chars.join(""); -} - -/** - * Force a length-preserving deterministic replacement to differ from the secret - * it stands in for. `generateDeterministicReplacement` seeds its first 1–2 chars - * with the `Z`/`ZZ` sentinel, so a whole configured value that is exactly `Z` or - * `ZZ` (or an astronomically unlikely longer hash collision) would otherwise be - * emitted unchanged and ship the raw secret to the provider. Flip the first char - * to a fixed different glyph: same length, still deterministic, guaranteed != the - * secret. Only safe for a whole CONFIGURED value (a plain secret matches its own - * literal, so the perturbed output is no longer matched and stays a fixed point); - * per-chunk remainders must keep the sentinel to remain idempotent across restart. - */ -function ensureDistinctReplacement(replacement: string, secret: string): string { - if (replacement.length === 0 || replacement !== secret) return replacement; - const alt = replacement[0] === REPLACEMENT_CHARS[0] ? REPLACEMENT_CHARS[1] : REPLACEMENT_CHARS[0]; - return alt + replacement.slice(1); -} - -// How far left of the matched span the re-match scan begins looking for a match -// that overlaps the candidate. This bounds ONLY the match-start search position, -// never the lookbehind/lookahead context: the probe below substitutes the -// candidate into the FULL text, so a regex's lookbehind/lookahead assertions -// always evaluate against complete context regardless of width. The single -// re-match this misses is one that begins more than this many bytes before the -// span and extends into it (a single match longer than the window) — that only -// churns the chosen redaction marker between candidates, never back to the raw -// matched value, so it cannot leak a secret. -const REGEX_REMATCH_BACKSCAN = 512; - -interface RegexMatchContext { - /** Full text the match was found in (positions are offsets into it). */ - text: string; - /** Start/end of the matched span being replaced. */ - start: number; - end: number; -} - -/** - * Whether `candidate`, substituted for the matched span in its surrounding text, - * is re-matched by `regex` at its own position. A replace-mode regex that depends - * on context (lookbehind/lookahead/`\b`) can match a candidate that does NOT match - * in isolation: e.g. `(?<=api=)[AZ]` never matches a bare `A`, but `api=A` does, so - * a candidate `A` chosen by an isolation test is re-redacted on the next obfuscate() - * pass and can oscillate back to the raw matched value. The probe substitutes the - * candidate into the FULL text — not a truncated window — so a wide lookbehind or - * lookahead (e.g. `(?<=A{600})`) still evaluates against the context that makes it - * match. Truncating that context dropped the assertion's reach and falsely - * accepted an oscillating, leaky candidate. The scan starts a bounded distance - * left of the span and stops once a match begins at/after the span's end (matches - * arrive in order), keeping per-candidate cost independent of total text length. - */ -function regexRematchesInContext(candidate: string, regex: RegExp, ctx: RegexMatchContext): boolean { - const probe = ctx.text.slice(0, ctx.start) + candidate + ctx.text.slice(ctx.end); - const spanStart = ctx.start; - const spanEnd = spanStart + candidate.length; - regex.lastIndex = Math.max(0, spanStart - REGEX_REMATCH_BACKSCAN); - for (let m = regex.exec(probe); m !== null; m = regex.exec(probe)) { - const matchStart = m.index; - const matchEnd = m.index + m[0].length; - // Matches arrive in increasing position; once one starts at or past the - // span's end it cannot cover the candidate, and neither can any later one. - if (matchStart >= spanEnd) break; - // A match overlapping the candidate's own bytes means those bytes get - // re-redacted on a later pass — not a fixed point. - if (matchEnd > spanStart) return true; - // Zero-width matches do not advance lastIndex; step past to avoid a loop. - if (m[0].length === 0) regex.lastIndex++; - } - return false; -} - -/** - * Search same-length replacements for one the regex does NOT match, so a default - * regex secret whose deterministic replacement collides with its own value (the - * `Z`/`ZZ` sentinel, or an astronomical hash collision) is still redacted to a - * STABLE nonmatching value instead of shipping the raw secret. A nonmatching - * candidate is a fixed point under re-obfuscation — the regex never re-matches it, - * so it cannot re-leak on a later pass. The search stays bounded to O(length * - * alphabet) regardless of value length: first exhaust every single-position - * substitution against a deterministic baseline (`AAAA…`, then `!AAA…`, `A!AA…`, - * …) so any regex that only needs one out-of-class byte — regardless of position — - * is found in a handful of probes rather than enumerating every combination (which - * for a 3-byte match-everything config, e.g. `[\s\S]{3}`, would otherwise run - * 90**3 = 729000 candidates through the regex on every single match, stalling - * provider requests). Candidates are enumerated deterministically over a stable - * ASCII alphabet: alphanumerics first (usually enough), then punctuation fallback - * bytes when the regex covers every alphanumeric candidate. When the regex still - * matches around a lone perturbed byte (for example `[A-Za-z0-9].*` matching the - * unperturbed tail), full-width same-byte candidates (`!!!!!`, `_____`, …) are - * tried next. When the regex covers every non-whitespace candidate (e.g. `\S{n}`), - * whitespace markers (a full space/tab run, then a single whitespace byte among - * non-whitespace filler) are tried as a last resort. A genuine match-everything - * regex (`.`/`[\s\S]`, which also matches space and tab) still exhausts this bounded - * sweep and returns undefined, letting the caller keep its own fixed-point fallback - * — bounded search can in principle miss an escape that depends jointly on - * multiple positions in a way no single-position swap reaches, but no realistic - * secret-redaction regex (character classes, literal matches, anchored/bounded - * repeats) has that shape. - */ -function findNonMatchingReplacement(value: string, regex: RegExp, context: RegexMatchContext): string | undefined { - const len = value.length; - if (len === 0) return undefined; - // Exhaust every single-position substitution against the deterministic baseline - // first (covers the common case cheaply), then fall back to full-width same-byte - // candidates for a regex that only rejects a lone perturbed byte in context. - const baseline = NONMATCHING_REPLACEMENT_CHARS[0].repeat(len); - for (let position = 0; position < len; position++) { - for (const ch of NONMATCHING_REPLACEMENT_CHARS) { - const candidate = `${baseline.slice(0, position)}${ch}${baseline.slice(position + 1)}`; - if (candidate === value) continue; - if (!regexRematchesInContext(candidate, regex, context)) return candidate; - } - } - // If the regex can still match around a lone punctuation byte (for example - // `[A-Za-z0-9].*` matching the `AAAA` tail of `!AAAA`), try full-width - // same-byte fallbacks like `!!!!!`, `_____`, etc. before giving up. - for (const ch of NONMATCHING_REPLACEMENT_CHARS) { - const candidate = ch.repeat(len); - if (candidate === value) continue; - if (!regexRematchesInContext(candidate, regex, context)) return candidate; - } - return findWhitespaceFallbackReplacement(value, regex, context); -} - -/** - * Last-resort fallback for a default replace regex that matches every - * non-whitespace candidate. Builds same-length whitespace markers the regex - * cannot match: first a full space/tab run (handles `\S`-class patterns), then a - * single whitespace byte among non-whitespace filler (` AAAA`, `A AAA`, …). The - * mixed marker defeats regexes that ALSO match all-space/all-tab runs, e.g. - * `(?:\S{n}| {n}|\t{n})`, because the lone whitespace byte breaks every - * fixed-length run. A genuine match-everything regex (`.`/`[\s\S]`) matches the - * filler and the whitespace alike, so this still returns undefined there, keeping - * the caller's sentinel as the sole fixed point. - */ -function findWhitespaceFallbackReplacement( - value: string, - regex: RegExp, - context: RegexMatchContext, -): string | undefined { - const len = value.length; - const filler = NONMATCHING_REPLACEMENT_CHARS[0]; - for (const ws of WHITESPACE_REPLACEMENT_CHARS) { - const full = ws.repeat(len); - if (full !== value) { - if (!regexRematchesInContext(full, regex, context)) return full; - } - for (let pos = 0; pos < len; pos++) { - const candidate = `${filler.repeat(pos)}${ws}${filler.repeat(len - pos - 1)}`; - if (candidate === value) continue; - if (!regexRematchesInContext(candidate, regex, context)) return candidate; - } - } - return undefined; -} - -/** - * Whether a default (no custom `replacement`) replace-mode regex can never - * safely redact a 1-2 char match: `findNonMatchingReplacement`'s bounded - * search — the same search `#generateRegexReplacement` runs at match time — - * finds no candidate the regex fails to re-match. This holds independent of - * any actual per-install key: the search already exhausts every character in - * `REPLACEMENT_CHARS` (the alphabet `buildKeyedReplacementRun` draws its - * fallback marker from) plus punctuation and whitespace, so if none of those - * escape the regex, no key-derived marker drawn from the same alphabet can - * either — the marker is guaranteed to re-match too, making every such match - * unresolvable: the fallback could only ever emit the raw matched text - * unchanged. Probed with a value (`"\0".repeat(length)`) the bounded search - * never treats as a real candidate, so the result depends only on the - * regex's own matching behavior, not on this specific probe. - */ -export function regexHasUnresolvableShortMatchFallback(regex: RegExp): boolean { - return ([1, 2] as const).some(length => { - const probe = "\u0000".repeat(length); - const savedLastIndex = regex.lastIndex; - try { - return findNonMatchingReplacement(probe, regex, { text: probe, start: 0, end: length }) === undefined; - } finally { - regex.lastIndex = savedLastIndex; - } - }); -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Placeholder format -// ═══════════════════════════════════════════════════════════════════════════ - -const HASH_CHARS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789"; -// Base length is sized for ~62 bits of entropy (64 bits of a keyed digest -// rendered as 12 base36 chars) so unrelated secrets do not collide on a shared -// base. A collision would let a persisted placeholder deobfuscate to the wrong -// secret when the configured secret set or its ordering changes across sessions. -const HASH_LEN = 12; -const MAX_FRIENDLY_NAME_LEN = 32; -// Plain/regex obfuscate matches shorter than this are toned down (never placed -// behind a reversible placeholder) to avoid redacting small words/fragments. -export const MIN_OBFUSCATE_SECRET_LEN = 8; - -// Per-process fallback key used when a caller does not supply a persisted -// per-install key. It is random (never shipped in source), so model-visible -// placeholders cannot be reversed by dictionary-hashing candidate secrets; it -// only forgoes cross-session token stability, which the persisted key provides. -let ephemeralPlaceholderKey: string | undefined; -function defaultPlaceholderKey(): string { - ephemeralPlaceholderKey ??= crypto.randomBytes(32).toString("base64url"); - return ephemeralPlaceholderKey; -} - -type PlaceholderCaseHint = "U" | "L" | "C" | "M"; - -/** Normalize a friendly name into the model-visible placeholder prefix. */ -export function sanitizeSecretFriendlyName(name: string): string | undefined { - const sanitized = name - .replace(/[^A-Za-z0-9]/g, "") - .toUpperCase() - .slice(0, MAX_FRIENDLY_NAME_LEN); - return sanitized.length > 0 ? sanitized : undefined; -} - -/** - * Normalize a secret value into the same alnum-only, uppercased shape a - * friendly-name label or placeholder prefix is sanitized into, so comparing a - * raw (possibly lowercase/punctuated) secret value against already-sanitized, - * model-visible text does not miss a case- or separator-only variant. Unlike - * `sanitizeSecretFriendlyName` this never truncates and never signals "empty" - * via `undefined` — callers already guard on `.length > 0` before comparing. - */ -function sanitizeForCollisionCheck(value: string): string { - return value.replace(/[^A-Za-z0-9]/g, "").toUpperCase(); -} - -// A label leaks a secret either by containing the whole normalized secret or, -// once it reaches the public display cap, by being the secret's visible prefix. -// Shorter names like "TOKEN" can still be intentional generic labels. -function sanitizedLabelCollidesWithSecret(sanitizedLabel: string, sanitizedSecret: string): boolean { - if (sanitizedSecret.length === 0) return false; - if (sanitizedLabel.includes(sanitizedSecret)) return true; - return sanitizedLabel.length >= MAX_FRIENDLY_NAME_LEN && sanitizedSecret.startsWith(sanitizedLabel); -} - -/** - * Whether an entry needs the persisted placeholder key: either because it can - * produce a reversible (keyed) obfuscate-mode placeholder, or because a default - * (no custom `replacement`) replace-mode regex can reach - * `#generateRegexReplacement`'s key-derived idempotent fallback marker (see - * `#generateReplacement`) when every same-length candidate re-matches a - * pathological match-everything config (e.g. `[\s\S]{8}`). That fallback depends - * on the persisted per-install key — not just length — to stay a fixed point - * across a process restart; without a persisted key, a fresh install falls back - * to a process-random key (`defaultPlaceholderKey()`), so the fallback marker - * would churn across restarts even though the algorithm itself is stable. A - * regex WITH a custom `replacement` never reaches that fallback (it always emits - * the literal configured string), and a plain replace secret's replacement is - * pure content-hash (`#generateSecretReplacement`), so neither needs the key. - * Short plain obfuscate entries are toned down (never placeheld), so they must - * NOT force key creation: otherwise a `secret-placeholder.key` file is written - * and persisted for a config that ends up with no active secrets, leaving the - * key readable via a tool and reusable for later placeholders. - */ -export function secretEntryNeedsPlaceholderKey(entry: SecretEntry): boolean { - if ((entry.mode ?? "obfuscate") === "obfuscate") { - if (entry.type === "regex") return true; - return entry.content.length >= MIN_OBFUSCATE_SECRET_LEN; - } - return entry.type === "regex" && entry.replacement === undefined; -} - -/** - * Whether a plain replace-mode replacement string can contribute a fragment that - * helps the replace phase reconstruct an obfuscate `content`. During obfuscate()'s - * replace phase the output is a tiling of passthrough bytes (adversary-controlled - * provider text) and whole replacement outputs; any contiguous occurrence of - * `content` in that output is covered by interior replacement tiles (each a - * substring of `content`) bordered by passthrough at the ends, where the border - * tile may be a suffix of a replacement (forming `content`'s prefix) or a prefix - * of a replacement (forming `content`'s suffix). An EMPTY replacement deletes its - * trigger entirely, joining the passthrough on both sides; with adversary-chosen - * surrounding bytes that can form any non-empty `content` across the deleted gap. - * So a replacement can help iff it is empty, is a substring of `content`, - * contains `content`, or shares such a border overlap. - */ -function replacementCanFormContent(replacement: string, content: string): boolean { - if (replacement.length === 0) return content.length > 0; - if (content.includes(replacement) || replacement.includes(content)) return true; - const maxOverlap = Math.min(replacement.length, content.length); - for (let k = 1; k <= maxOverlap; k++) { - // A suffix of the replacement forms the prefix of the content (left border), - // or a prefix of the replacement forms the suffix of the content (right border). - if (content.startsWith(replacement.slice(replacement.length - k)) || content.endsWith(replacement.slice(0, k))) { - return true; - } - } - return false; -} - -/** - * Whether a SET of entries needs the persisted placeholder key. `obfuscate()` - * applies plain replace-mode mappings before the plain-obfuscate pass, so a plain - * obfuscate entry only emits a reversible (keyed) placeholder when its content can - * still appear AFTER the replace phase. When no obfuscate entry can ever produce a - * placeholder, the persisted key must NOT be required/created — otherwise an - * effectively replace-only secret set still writes `secret-placeholder.key` and - * fails startup when the agent config dir is unwritable. - * - * The decision models the replace phase as the obfuscator actually runs it: - * replace mappings are content-keyed (later duplicate wins) and applied in - * descending content-length order; for a fresh probe (no prior placeholders) that - * phase is plain sequential substring replacement. A plain obfuscate entry needs - * the key when its content survives that simulated phase (direct typing) OR when - * any effective replacement can form the content via tiling — a substring, - * wholesale superstring, or prefix/suffix border that joins with surrounding - * passthrough bytes (see `replacementCanFormContent`). This covers direct - * shadowing (`SECRET -> safe`), reintroduction, duplicate ordering, transitive - * chains, and context-joined fragments uniformly. Default (omitted) replacements - * are deterministic, length-preserving, and distinct, so a same-content shadow - * with no other interacting replacement stays key-free. - * Replacement outputs are themselves rewritten by every later (shorter-content) - * replacement before the plain-obfuscate pass sees them, so a fragment that a - * subsequent replacement erases (`AA -> SEC` then `S -> X` turns every `SEC` into - * `XEC`) no longer forces the key. Surrounding bytes stay modeled as arbitrary - * passthrough, so testing the surviving fragment only drops false positives and - * never under-approximates a real key need. - */ -export function secretEntriesNeedPlaceholderKey(entries: SecretEntry[]): boolean { - const replaceMap = new Map(); - for (const entry of entries) { - if (entry.type !== "plain" || (entry.mode ?? "obfuscate") !== "replace") continue; - replaceMap.set( - entry.content, - entry.replacement ?? ensureDistinctReplacement(generateDeterministicReplacement(entry.content), entry.content), - ); - } - const replacePhase = [...replaceMap].sort((a, b) => b[0].length - a[0].length); - // Apply the replace phase from `start` onward. The phase runs in descending - // content-length order, so a replacement output emitted at index i is rewritten - // only by the later (shorter-content) replacements at i+1…; `start` 0 models a - // value typed directly into the input. - const applyReplacePhaseFrom = (text: string, start: number): string => { - let result = text; - for (let i = start; i < replacePhase.length; i++) { - result = result.split(replacePhase[i][0]).join(replacePhase[i][1]); - } - return result; - }; - return entries.some(entry => { - if (!secretEntryNeedsPlaceholderKey(entry)) return false; - // Regex obfuscate entries match dynamically; conservatively require the key. - if (entry.type !== "plain") return true; - const content = entry.content; - if (applyReplacePhaseFrom(content, 0).includes(content)) return true; - // Test each replacement output in the form it SURVIVES the rest of the phase, - // so a fragment a later replacement erases no longer forces the key. The - // content it tiles into must also survive those later replacements: if a - // shorter-content replacement rewrites the surrounding passthrough bytes - // (e.g. `AA -> SEC` forms `SEC`+`RET12`, then `R -> X` turns the freshly - // formed `SECRET12` into `SECXET12`), the content can never reach the - // obfuscate pass, so the key is not needed. Requiring content stability only - // drops such false positives — a formation that genuinely survives is still - // caught at the replacement index that produces it. - return replacePhase.some( - ([, replacement], i) => - applyReplacePhaseFrom(content, i + 1) === content && - replacementCanFormContent(applyReplacePhaseFrom(replacement, i + 1), content), - ); - }); -} - -// Derive the model-visible base from a KEYED digest of the secret. xxHash is -// fast and unkeyed, so a fixed-seed content hash of a low-entropy secret could -// be dictionaried from the transcript; HMAC-SHA256 under a private per-install -// key cannot, since the attacker lacks the key. -function buildHashBase(key: string, value: string): string { - const digest = new Bun.CryptoHasher("sha256", key).update(value).digest(); - let v = 0n; - for (let i = 0; i < 8; i++) v = (v << 8n) | BigInt(digest[i]); - const radix = BigInt(HASH_CHARS.length); - let tag = ""; - for (let i = 0; i < HASH_LEN; i++) { - tag += HASH_CHARS[Number(v % radix)]; - v /= radix; - } - return tag; -} - -// Build a deterministic, key-derived run of REPLACEMENT_CHARS of the given -// length. Used to redact a per-chunk replace remainder to a marker that depends -// only on the per-install key and the remainder length, so a fresh obfuscator -// reproduces the identical marker (idempotent redaction across restarts) while -// the run stays unpredictable without the key (raw sentinel-shaped bytes cannot -// equal the marker, so they are still redacted rather than passed through). -function buildKeyedReplacementRun(key: string, length: number): string { - if (length <= 0) return ""; - const radix = REPLACEMENT_CHARS.length; - let out = ""; - for (let block = 0; out.length < length; block++) { - const digest = new Bun.CryptoHasher("sha256", key).update(`replace-chunk\0${length}\0${block}`).digest(); - for (let i = 0; i < digest.length && out.length < length; i++) { - out += REPLACEMENT_CHARS[digest[i] % radix]; - } - } - return out; -} - -function inferCaseHint(secret: string): PlaceholderCaseHint | undefined { - let hasCased = false; - let hasUpper = false; - let hasLower = false; - let capitalized = true; - let seenFirstCased = false; - - for (let i = 0; i < secret.length; i++) { - const code = secret.charCodeAt(i); - const isUpper = code >= 65 && code <= 90; - const isLower = code >= 97 && code <= 122; - if (!isUpper && !isLower) continue; - - hasCased = true; - if (isUpper) { - hasUpper = true; - if (seenFirstCased) capitalized = false; - } else { - hasLower = true; - if (!seenFirstCased) capitalized = false; - } - seenFirstCased = true; - } - - if (!hasCased) return undefined; - if (hasUpper && !hasLower) return "U"; - if (hasLower && !hasUpper) return "L"; - if (capitalized) return "C"; - return "M"; -} - -function buildPlaceholder(hint: PlaceholderCaseHint | undefined, base: string, friendlyName?: string): string { - const prefix = friendlyName ? `${friendlyName}_` : ""; - return hint ? `$$${prefix}${base}:${hint}$$` : `$$${prefix}${base}$$`; -} - -/** Regex matching `$$HASH$$`, `$$HASH:U$$`, and `$$FRIENDLY_HASH(:hint)$$` placeholders. */ -const PLACEHOLDER_RE = /\$\$(?:[A-Z0-9]+_)?[A-Z0-9]{4,}(?::[ULCM])?\$\$/g; - -function resumePlaceholderScanAfterRejectedCandidate(match: RegExpExecArray): void { - // RegExp#exec does not find overlapping matches. Restart at the rejected - // candidate's closing delimiter, which can open an immediately adjacent placeholder. - PLACEHOLDER_RE.lastIndex = match.index + match[0].length - 2; -} - -function placeholderWithoutFriendlyName(placeholder: string): string | undefined { - const match = /^\$\$[A-Z0-9]+_([A-Z0-9]{4,}(?::[ULCM])?)\$\$$/.exec(placeholder); - return match ? `$$${match[1]}$$` : undefined; -} - -function lookupFriendlyPlaceholderAlias( - deobfuscateMap: ReadonlyMap, - placeholder: string, -): { secret: string; recursive: boolean } | undefined { - const direct = deobfuscateMap.get(placeholder); - if (direct !== undefined) return direct; - const unprefixed = placeholderWithoutFriendlyName(placeholder); - return unprefixed !== undefined ? deobfuscateMap.get(unprefixed) : undefined; -} - -const PENDING_PLACEHOLDER_SUFFIX_RE = /(?:\$\$(?:[A-Z0-9]+_)?[A-Z0-9]*(?::[ULCM]?)?|\$)$/; - -// Withhold a trailing run that could be the start of a placeholder from streamed -// deltas, so a partial token is never emitted before deobfuscation can replace -// it. A lone trailing delimiter character is always buffered because it can open -// a placeholder; the final non-streamed flush re-emits it when no token follows. -export function stripPendingSecretPlaceholderSuffix(text: string): string { - const pendingPlaceholderStart = text.match(PENDING_PLACEHOLDER_SUFFIX_RE); - if (pendingPlaceholderStart?.index === undefined) return text; - return text.slice(0, pendingPlaceholderStart.index); -} - -interface RegexScanSegment { - scanStart: number; - scanEnd: number; - textStart: number; - textEnd: number; - generatedPlaceholder: boolean; - recursive: boolean; -} - -interface ReplaceRegexScan { - text: string; - segments: RegexScanSegment[]; -} - // ═══════════════════════════════════════════════════════════════════════════ // SecretObfuscator // ═══════════════════════════════════════════════════════════════════════════ @@ -1847,783 +1363,3 @@ export class SecretObfuscator { return matches.reverse(); } } - -// ═══════════════════════════════════════════════════════════════════════════ -// Display restore (inbound, persisted/provider → local display) -// ═══════════════════════════════════════════════════════════════════════════ - -/** - * Restore secret placeholders for local display. Only message kinds the model - * itself authored from obfuscated context carry placeholders — assistant - * content and the LLM-written branch/compaction summaries. User, developer, and - * tool-result messages are persisted with their literal text, so operator-authored - * placeholder-shaped text must survive untouched; those roles are never walked. - */ -export function deobfuscateSessionContext( - sessionContext: SessionContext, - obfuscator: SecretObfuscator | undefined, -): SessionContext { - if (!obfuscator?.hasSecrets()) return sessionContext; - const messages = deobfuscateAgentMessages(obfuscator, sessionContext.messages); - return messages === sessionContext.messages ? sessionContext : { ...sessionContext, messages }; -} - -export function deobfuscateAgentMessages(obfuscator: SecretObfuscator, messages: AgentMessage[]): AgentMessage[] { - const deob = (text: string): string => obfuscator.deobfuscate(text); - let changed = false; - const result = messages.map((message): AgentMessage => { - switch (message.role) { - case "assistant": { - const content = deobfuscateAssistantContent(obfuscator, message.content); - if (content === message.content) return message; - changed = true; - return { ...message, content }; - } - case "branchSummary": { - const summary = deob(message.summary); - if (summary === message.summary) return message; - changed = true; - return { ...message, summary }; - } - case "compactionSummary": { - const summary = deob(message.summary); - const shortSummary = message.shortSummary === undefined ? undefined : deob(message.shortSummary); - const blocks = message.blocks === undefined ? undefined : deobfuscateTextBlocks(obfuscator, message.blocks); - if (summary === message.summary && shortSummary === message.shortSummary && blocks === message.blocks) { - return message; - } - changed = true; - return { ...message, summary, shortSummary, blocks }; - } - default: - return message; - } - }); - return changed ? result : messages; -} - -/** - * Restore placeholders in assistant content: visible text and tool-call - * arguments/intent/rawBlock. Thinking and signatures are opaque - * provider-replay/hidden-reasoning data and pass through byte-identical. - */ -export function deobfuscateAssistantContent( - obfuscator: SecretObfuscator, - content: AssistantMessage["content"], -): AssistantMessage["content"] { - if (!obfuscator.hasSecrets()) return content; - const deob = (text: string): string => obfuscator.deobfuscate(text); - let changed = false; - const result = content.map((block): AssistantMessage["content"][number] => { - if (block.type === "text") { - const text = deob(block.text); - if (text === block.text) return block; - changed = true; - return { ...block, text }; - } - - if (block.type === "toolCall") { - const args = deobfuscateToolArguments(obfuscator, block.arguments); - const intent = block.intent === undefined ? undefined : deob(block.intent); - const rawBlock = block.rawBlock === undefined ? undefined : deob(block.rawBlock); - if (args === block.arguments && intent === block.intent && rawBlock === block.rawBlock) return block; - changed = true; - return { ...block, arguments: args, intent, rawBlock }; - } - return block; - }); - return changed ? result : content; -} - -/** - * Restore placeholders inside a tool call's arguments. Arguments are arbitrary - * model-authored JSON, so tool-call arguments are the ONLY place a recursive - * JSON walk runs. - */ -export function deobfuscateToolArguments( - obfuscator: SecretObfuscator, - args: Record, -): Record { - if (!obfuscator.hasSecrets()) return args; - return mapJsonStrings(args as JsonValue, s => obfuscator.deobfuscate(s)) as Record; -} - -/** Redact secrets inside a tool call's arguments (same JSON-walk exception as {@link deobfuscateToolArguments}). */ -export function obfuscateToolArguments( - obfuscator: SecretObfuscator, - args: Record, - sharedRegexSecretValues?: ReadonlySet, -): Record { - if (!obfuscator.hasSecrets()) return args; - const regexSecretValues = sharedRegexSecretValues ?? collectJsonRegexSecretValues(obfuscator, args as JsonValue); - return mapJsonStrings(args as JsonValue, s => obfuscator.obfuscate(s, regexSecretValues)) as Record; -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Outbound obfuscation (local → provider) -// ═══════════════════════════════════════════════════════════════════════════ - -type UserFacingMessage = Extract; - -/** Obfuscate `text` blocks of a content array; image and other blocks pass through. */ -function obfuscateTextBlocks( - obfuscator: SecretObfuscator, - content: (TextContent | ImageContent)[], - sharedRegexSecretValues?: ReadonlySet, -): (TextContent | ImageContent)[] { - let changed = false; - const result = content.map((block): TextContent | ImageContent => { - if (block.type !== "text") return block; - const text = obfuscator.obfuscate(block.text, sharedRegexSecretValues); - if (text === block.text) return block; - changed = true; - return { ...block, text }; - }); - return changed ? result : content; -} - -/** Restore placeholders in `text` blocks of a content array; image and other blocks pass through. */ -function deobfuscateTextBlocks( - obfuscator: SecretObfuscator, - content: (TextContent | ImageContent)[], -): (TextContent | ImageContent)[] { - let changed = false; - const result = content.map((block): TextContent | ImageContent => { - if (block.type !== "text") return block; - const text = obfuscator.deobfuscate(block.text); - if (text === block.text) return block; - changed = true; - return { ...block, text }; - }); - return changed ? result : content; -} - -/** - * Re-obfuscate assistant content before it returns to a provider after session - * restoration, removing friendly prefixes made unsafe by this batch. A changed - * thinking block loses its byte-bound replay signature. - */ -function obfuscateAssistantContentForReplay( - obfuscator: SecretObfuscator, - content: AssistantMessage["content"], - sharedRegexSecretValues: ReadonlySet, -): AssistantMessage["content"] { - const obfuscate = (text: string): string => - obfuscator.stripUnsafeFriendlyPlaceholderPrefixes( - obfuscator.obfuscate(text, sharedRegexSecretValues), - sharedRegexSecretValues, - ); - let changed = false; - const result = content.map((block): AssistantMessage["content"][number] => { - if (block.type === "text") { - const text = obfuscate(block.text); - if (text === block.text) return block; - changed = true; - return { ...block, text }; - } - if (block.type === "thinking") { - const thinking = obfuscate(block.thinking); - if (thinking === block.thinking) return block; - changed = true; - return { ...block, thinking, thinkingSignature: undefined }; - } - if (block.type === "toolCall") { - const args = mapJsonStrings(block.arguments as JsonValue, obfuscate) as Record; - const intent = block.intent === undefined ? undefined : obfuscate(block.intent); - const rawBlock = block.rawBlock === undefined ? undefined : obfuscate(block.rawBlock); - if (args === block.arguments && intent === block.intent && rawBlock === block.rawBlock) return block; - changed = true; - return { ...block, arguments: args, intent, rawBlock }; - } - return block; - }); - return changed ? result : content; -} - -function collectMessageRegexSecretValues(obfuscator: SecretObfuscator, messages: Message[]): Set { - const values = new Set(); - const addText = (text: string | undefined): void => { - if (text === undefined) return; - for (const value of obfuscator.collectRegexSecretValuesForObfuscation(text)) { - values.add(value); - } - }; - for (const message of messages) { - if (message.role === "assistant") { - for (const block of message.content) { - if (block.type === "text") addText(block.text); - else if (block.type === "thinking") addText(block.thinking); - else if (block.type === "toolCall") { - for (const value of collectJsonRegexSecretValues(obfuscator, block.arguments as JsonValue)) { - values.add(value); - } - addText(block.intent); - addText(block.rawBlock); - } - } - continue; - } - if ( - message.role !== "user" && - message.role !== "toolResult" && - !(message.role === "developer" && message.attribution === "user") - ) { - continue; - } - const target = message as UserFacingMessage; - if (typeof target.content === "string") { - addText(target.content); - continue; - } - for (const block of target.content) { - if (block.type === "text") addText(block.text); - } - } - return values; -} - -/** - * Redact secrets from outbound messages. User messages, tool results, and - * user-authored developer messages (e.g. `@file` mentions) are obfuscated. - * Assistant replay content is re-obfuscated too, because session restoration - * expands keyed placeholders locally before the next provider request. Inline - * image bytes are never walked. - */ -export function obfuscateMessages(obfuscator: SecretObfuscator, messages: Message[]): Message[] { - if (!obfuscator.hasSecrets()) return messages; - const sharedRegexSecretValues = collectMessageRegexSecretValues(obfuscator, messages); - let changed = false; - const result = messages.map((message): Message => { - if ( - message.role !== "user" && - message.role !== "toolResult" && - !(message.role === "developer" && message.attribution === "user") - ) { - if (message.role !== "assistant") return message; - const content = obfuscateAssistantContentForReplay(obfuscator, message.content, sharedRegexSecretValues); - if (content === message.content) return message; - changed = true; - return { ...message, content }; - } - const target = message as UserFacingMessage; - if (typeof target.content === "string") { - const content = obfuscator.obfuscate(target.content, sharedRegexSecretValues); - if (content === target.content) return message; - changed = true; - return { ...target, content } as Message; - } - const content = obfuscateTextBlocks(obfuscator, target.content, sharedRegexSecretValues); - if (content === target.content) return message; - changed = true; - return { ...target, content } as Message; - }); - return changed ? result : messages; -} - -/** - * Redact outbound provider context. Only conversation messages are rewritten; - * the static system prompt and tool schemas pass through unchanged. - */ -export function obfuscateProviderContext(obfuscator: SecretObfuscator | undefined, context: Context): Context { - if (!obfuscator?.hasSecrets()) return context; - const messages = obfuscateMessages(obfuscator, context.messages); - return messages === context.messages ? context : { ...context, messages }; -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Helpers -// ═══════════════════════════════════════════════════════════════════════════ - -// Like the untracked walk, but threads a parallel `origin` tag string through: -// preserved placeholder spans keep their existing origin tag (so a -// same-call-fresh "F" placeholder is never relabeled prior-call "I", and vice -// versa), while `transform`'s output — always freshly generated or redacted -// content in both callers below — is tagged "I" (it must not be re-matched as -// though it arrived in the input, mirroring plain-secret replacement tagging). -function transformOutsidePlaceholdersTracked( - text: string, - origin: string, - shouldSkipPlaceholder: (placeholder: string) => boolean, - transform: (chunk: string) => string, - preservePlaceholder?: (placeholder: string) => string, -): { text: string; origin: string } { - PLACEHOLDER_RE.lastIndex = 0; - let result = ""; - let resultOrigin = ""; - let pendingIndex = 0; - for (;;) { - const match = PLACEHOLDER_RE.exec(text); - if (match === null) break; - if (!shouldSkipPlaceholder(match[0])) { - resumePlaceholderScanAfterRejectedCandidate(match); - continue; - } - const transformed = transform(text.slice(pendingIndex, match.index)); - result += transformed; - resultOrigin += "I".repeat(transformed.length); - const preserved = preservePlaceholder ? preservePlaceholder(match[0]) : match[0]; - result += preserved; - resultOrigin += origin.slice(match.index, match.index + match[0].length); - pendingIndex = match.index + match[0].length; - } - const trailing = transform(text.slice(pendingIndex)); - result += trailing; - resultOrigin += "I".repeat(trailing.length); - return { text: result, origin: resultOrigin }; -} - -function trailingOutsidePreservedPlaceholderChunk( - text: string, - shouldPreservePlaceholder: (placeholder: string) => boolean, -): string { - PLACEHOLDER_RE.lastIndex = 0; - let pendingIndex = 0; - let sawPlaceholder = false; - for (;;) { - const match = PLACEHOLDER_RE.exec(text); - if (match === null) break; - if (!shouldPreservePlaceholder(match[0])) { - resumePlaceholderScanAfterRejectedCandidate(match); - continue; - } - sawPlaceholder = true; - pendingIndex = match.index + match[0].length; - } - return sawPlaceholder ? text.slice(pendingIndex) : ""; -} - -function buildReplaceRegexScan( - text: string, - ranges: ReadonlyArray<{ start: number; end: number }>, - deobfuscateMap: ReadonlyMap, -): ReplaceRegexScan { - let scanText = ""; - let cursor = 0; - const segments: RegexScanSegment[] = []; - const appendSegment = ( - value: string, - textStart: number, - textEnd: number, - generatedPlaceholder: boolean, - recursive: boolean, - ) => { - if (value.length === 0) return; - const scanStart = scanText.length; - scanText += value; - segments.push({ - scanStart, - scanEnd: scanStart + value.length, - textStart, - textEnd, - generatedPlaceholder, - recursive, - }); - }; - - for (const range of ranges) { - appendSegment(text.slice(cursor, range.start), cursor, range.start, false, false); - const placeholder = text.slice(range.start, range.end); - const mapping = lookupFriendlyPlaceholderAlias(deobfuscateMap, placeholder); - appendSegment(mapping?.secret ?? placeholder, range.start, range.end, true, mapping?.recursive ?? false); - cursor = range.end; - } - appendSegment(text.slice(cursor), cursor, text.length, false, false); - - return { text: scanText, segments }; -} - -function mapReplaceRegexMatch( - segments: ReadonlyArray, - scanStart: number, - scanEnd: number, -): { - start: number; - end: number; - recursive: boolean; - preserveGeneratedPlaceholders: boolean; - partialPlaceholderCut: boolean; - cutResumeIndex: number; - firstPlaceholderScanStart: number; -} { - const startSegment = findScanSegment(segments, scanStart); - const endSegment = findScanSegment(segments, scanEnd - 1); - const start = startSegment.generatedPlaceholder - ? startSegment.textStart - : startSegment.textStart + (scanStart - startSegment.scanStart); - const end = endSegment.generatedPlaceholder - ? endSegment.textEnd - : endSegment.textStart + (scanEnd - endSegment.scanStart); - // A match boundary that falls strictly inside a generated placeholder's - // expanded value cuts the underlying secret: the snap above pulls the span out - // to the whole `#…#` token, so the obfuscate path can leave it alone instead of - // consuming a partial placeholder expansion. - const partialPlaceholderCut = - (startSegment.generatedPlaceholder && scanStart > startSegment.scanStart) || - (endSegment.generatedPlaceholder && scanEnd < endSegment.scanEnd); - let recursive = false; - let preserveGeneratedPlaceholders = false; - // When the match straddles a placeholder, resume scanning just past the last - // overlapping placeholder so trailing wholly-outside content (e.g. an 8-char - // run after the secret) still gets matched instead of being consumed by the - // straddling span. `firstPlaceholderScanStart` marks where the leading - // wholly-outside prefix ends, so a prefix that independently matches can be - // redacted on its own rather than skipped along with the cut span. - let cutResumeIndex = scanStart; - let firstPlaceholderScanStart = -1; - for (const segment of segments) { - if (segment.scanStart >= scanEnd || segment.scanEnd <= scanStart) continue; - recursive ||= segment.recursive; - preserveGeneratedPlaceholders ||= segment.generatedPlaceholder; - if (segment.generatedPlaceholder) { - if (firstPlaceholderScanStart === -1) firstPlaceholderScanStart = segment.scanStart; - if (segment.scanEnd > cutResumeIndex) cutResumeIndex = segment.scanEnd; - } - } - return { - start, - end, - recursive, - preserveGeneratedPlaceholders, - partialPlaceholderCut, - cutResumeIndex, - firstPlaceholderScanStart, - }; -} - -function findScanSegment(segments: ReadonlyArray, scanIndex: number): RegexScanSegment { - for (const segment of segments) { - if (scanIndex >= segment.scanStart && scanIndex < segment.scanEnd) return segment; - } - throw new Error("regex match did not map to source text"); -} - -/** - * Extend a scan-space resume position past a consecutive run of generated - * placeholder segments starting exactly at it, with no raw gap in between. A - * cut-resolution resume point that happens to land precisely on the START of - * ANOTHER placeholder must not stop there and hand it to a fresh `regex.exec` - * attempt — the same content, scanned as an opaque adjacent placeholder run, - * must resolve identically whether the run's LEADING member is still raw text - * (this call is about to placeholder it) or is ALREADY a placeholder from a - * prior call or an earlier pass of this same call. Without this, a bounded - * regex whose reach spans two adjacent secrets plus trailing spillover bytes - * (e.g. `[A-Z]{9}` over `ABCDEFGH` + `SECRETUV` + `A`) resolves the leading - * secret as its own independent redaction on the FIRST obfuscate() call (a - * genuinely raw prefix gets its own match, then the discard for the rest - * resumes right after it), but on a LATER call — once that prefix is itself a - * placeholder — the very first match attempt starts already inside the - * placeholder run, cannot be prefix-narrowed at all, and its discard resume - * point lands mid-run instead of past it, exposing a shorter tail (`SECRETUV` - * + `A`) to a clean, un-cut match the first call never attempted. Chaining the - * resume point through every immediately-adjacent placeholder makes both - * calls land on the exact same next scan position. - */ -function extendPastAdjacentPlaceholders(segments: ReadonlyArray, index: number): number { - let cursor = index; - for (;;) { - const segment = segments.find(candidate => candidate.scanStart === cursor && candidate.generatedPlaceholder); - if (!segment) return cursor; - cursor = segment.scanEnd; - } -} - -// Apply a fixed custom replacement across a matched span while preserving any -// inner generated placeholders. Usually the replacement is the user's single -// redaction marker for the whole match, so emit it for the first non-empty -// surrounding chunk and drop later chunks. But bounded regexes can cut through -// an already-emitted marker on the trailing side (`X#…#RED` from -// `XSECRETUVREDACTED`), where dropping the later prefix would leave raw bytes -// (`ACTED`) to be consumed on the next pass. Promote later chunks that are a -// prefix of the replacement to the FULL marker so the first pass is already a -// fixed point. The reversible placeholder stays intact in its relative -// position. -function redactWithFixedReplacementOutsidePlaceholders( - text: string, - origin: string, - replacement: string, - shouldPreservePlaceholder: (placeholder: string) => boolean, -): { text: string; origin: string } { - let emitted = false; - return transformOutsidePlaceholdersTracked( - text, - origin, - shouldPreservePlaceholder, - chunk => { - if (chunk.length === 0) return ""; - if (!emitted) { - emitted = true; - return replacement; - } - return replacement.startsWith(chunk) ? replacement : ""; - }, - placeholder => placeholder, - ); -} - -function deobfuscateGeneratedPlaceholderRanges( - text: string, - start: number, - end: number, - ranges: ReadonlyArray<{ start: number; end: number }>, - deobfuscateMap: ReadonlyMap, -): { text: string; recursive: boolean } { - let result = ""; - let cursor = start; - let recursive = false; - for (const range of ranges) { - if (range.end <= start || range.start >= end) continue; - const overlapStart = Math.max(range.start, start); - const overlapEnd = Math.min(range.end, end); - result += text.slice(cursor, overlapStart); - const placeholder = text.slice(overlapStart, overlapEnd); - const mapping = lookupFriendlyPlaceholderAlias(deobfuscateMap, placeholder); - result += mapping?.secret ?? placeholder; - recursive ||= mapping?.recursive ?? false; - cursor = overlapEnd; - } - result += text.slice(cursor, end); - return { text: result, recursive }; -} - -// Concatenate ONLY the deobfuscated placeholder ranges within [start, end), -// dropping the bytes that lie outside them. Used to test whether a regex match -// that straddles a prior-call placeholder would still match on the placeholder's -// own (expanded) secret value alone — i.e. the surrounding raw bytes are greedy -// spillover the match does not need, rather than content the match depends on. -function placeholderInnerText( - text: string, - start: number, - end: number, - ranges: ReadonlyArray<{ start: number; end: number }>, - deobfuscateMap: ReadonlyMap, -): string { - let result = ""; - for (const range of ranges) { - if (range.end <= start || range.start >= end) continue; - const overlapStart = Math.max(range.start, start); - const overlapEnd = Math.min(range.end, end); - const placeholder = text.slice(overlapStart, overlapEnd); - const mapping = lookupFriendlyPlaceholderAlias(deobfuscateMap, placeholder); - result += mapping?.secret ?? placeholder; - } - return result; -} - -// Concatenate the bytes of [start, end) that lie OUTSIDE the given (ascending, -// non-overlapping) placeholder ranges. Used to test whether a regex match that -// straddles a prior-call placeholder would still match on its surrounding bytes -// alone — i.e. those bytes are genuinely new content to redact rather than a -// match that only exists because the deobfuscated placeholder bridges them. -function textOutsidePlaceholderRanges( - text: string, - start: number, - end: number, - ranges: ReadonlyArray<{ start: number; end: number }>, -): string { - let result = ""; - let cursor = start; - for (const range of ranges) { - if (range.end <= start || range.start >= end) continue; - const overlapStart = Math.max(range.start, start); - const overlapEnd = Math.min(range.end, end); - result += text.slice(cursor, overlapStart); - cursor = overlapEnd; - } - result += text.slice(cursor, end); - return result; -} - -// Like `textOutsidePlaceholderRanges`, but tests each outside chunk against -// `regex` in its REAL context instead of on an isolated slice — tried in BOTH -// the literal `#…#` placeholder-token text AND the EXPANDED scan context -// (placeholder resolved to its secret value), since either can be the reason a -// chunk independently requires redaction: -// - Literal-token context matters when the placeholder TOKEN's own non-word -// boundary is what completes a boundary-sensitive pattern, e.g. a prefix -// "ABCDEFGH" next to a placeholder token matches `\b[A-Z]{8}\b` because the -// token's leading `#` is a non-word byte — but that boundary disappears -// once the placeholder expands into more `[A-Z]` bytes with no separator. -// - Expanded scan context matters when a lookbehind/lookahead only resolves -// once the neighboring placeholder is expanded, e.g. a prior plain -// placeholder for `ABCDEFGH` next to raw `SECRET`, matched by -// `(?<=ABCDEFGH)SECRET`: the literal placeholder token before `SECRET` -// never satisfies the lookbehind, so literal-context alone wrongly reports -// no independent match. -// A match only counts when it lies ENTIRELY within one outside chunk (in -// whichever context it was tested); a match that reaches into the -// placeholder itself is not evidence the outside chunk independently -// requires redaction. -function outsidePlaceholderRangesAnyIndependentlyMatch( - text: string, - scanText: string, - segments: ReadonlyArray, - start: number, - end: number, - ranges: ReadonlyArray<{ start: number; end: number }>, - regex: RegExp, -): boolean { - // A text-space outside chunk lies entirely within one non-placeholder scan - // segment (placeholder ranges are exactly the gaps between such segments), - // so its scan-space span is a fixed offset from its text-space span. - const toScanSpace = (chunkStart: number, chunkEnd: number): [number, number] | undefined => { - for (const segment of segments) { - if (segment.generatedPlaceholder || segment.textStart > chunkStart || segment.textEnd < chunkEnd) continue; - const offset = segment.scanStart - segment.textStart; - return [chunkStart + offset, chunkEnd + offset]; - } - return undefined; - }; - const chunkIndependentlyMatches = (chunkStart: number, chunkEnd: number): boolean => { - if (chunkMatchesInSourceContext(text, chunkStart, chunkEnd, regex)) return true; - const scanSpan = toScanSpace(chunkStart, chunkEnd); - return scanSpan !== undefined && chunkMatchesInSourceContext(scanText, scanSpan[0], scanSpan[1], regex); - }; - let cursor = start; - for (const range of ranges) { - if (range.end <= start || range.start >= end) continue; - const overlapStart = Math.max(range.start, start); - const overlapEnd = Math.min(range.end, end); - if (cursor < overlapStart && chunkIndependentlyMatches(cursor, overlapStart)) return true; - cursor = overlapEnd; - } - return cursor < end && chunkIndependentlyMatches(cursor, end); -} - -// Whether `regex` (global) has a match fully contained in [chunkStart, chunkEnd) -// when run against the full `text` — so lookbehind/lookahead see the actual -// surrounding bytes rather than an isolated slice's edges. -function chunkMatchesInSourceContext(text: string, chunkStart: number, chunkEnd: number, regex: RegExp): boolean { - regex.lastIndex = chunkStart; - for (;;) { - const found = regex.exec(text); - if (found === null || found.index >= chunkEnd) return false; - const matchEnd = found.index + found[0].length; - if (matchEnd <= chunkEnd) return true; - regex.lastIndex = found[0].length === 0 ? found.index + 1 : matchEnd; - } -} - -function firstOutsidePlaceholderRange( - start: number, - end: number, - ranges: ReadonlyArray<{ start: number; end: number }>, -): { start: number; end: number } | undefined { - let cursor = start; - for (const range of ranges) { - if (range.end <= start || range.start >= end) continue; - const overlapStart = Math.max(range.start, start); - const overlapEnd = Math.min(range.end, end); - if (cursor < overlapStart) return { start: cursor, end: overlapStart }; - cursor = overlapEnd; - } - return cursor < end ? { start: cursor, end } : undefined; -} - -function countOutsidePlaceholderRanges( - start: number, - end: number, - ranges: ReadonlyArray<{ start: number; end: number }>, -): number { - let count = 0; - let cursor = start; - for (const range of ranges) { - if (range.end <= start || range.start >= end) continue; - const overlapStart = Math.max(range.start, start); - const overlapEnd = Math.min(range.end, end); - if (cursor < overlapStart) count++; - cursor = overlapEnd; - } - if (cursor < end) count++; - return count; -} - -function replaceRange(text: string, start: number, end: number, replacement: string): string { - return text.slice(0, start) + replacement + text.slice(end); -} - -/** Deep-walk an object, transforming all string values. */ -function deepWalkStrings(obj: T, transform: (s: string) => string): T { - if (typeof obj === "string") { - return transform(obj) as unknown as T; - } - if (Array.isArray(obj)) { - let changed = false; - const result = obj.map(item => { - const transformed = deepWalkStrings(item, transform); - if (transformed !== item) changed = true; - return transformed; - }); - return (changed ? result : obj) as unknown as T; - } - if (obj !== null && typeof obj === "object" && isPlainRecord(obj)) { - let changed = false; - const result: Record = {}; - for (const key of Object.keys(obj)) { - const value = (obj as Record)[key]; - const transformed = deepWalkStrings(value, transform); - if (transformed !== value) changed = true; - result[key] = transformed; - } - return (changed ? result : obj) as T; - } - return obj; -} - -function isPlainRecord(obj: object): obj is Record { - const prototype = Object.getPrototypeOf(obj); - return prototype === Object.prototype || prototype === null; -} - -function collectJsonRegexSecretValues(obfuscator: SecretObfuscator, value: JsonValue): Set { - const values = new Set(); - const collect = (item: JsonValue): void => { - if (typeof item === "string") { - for (const secretValue of obfuscator.collectRegexSecretValuesForObfuscation(item)) { - values.add(secretValue); - } - return; - } - if (Array.isArray(item)) { - for (const child of item) collect(child); - return; - } - if (item !== null && typeof item === "object") { - for (const child of Object.values(item)) { - if (child !== undefined) collect(child); - } - } - }; - collect(value); - return values; -} - -/** - * Map every string in arbitrary JSON. Used ONLY for tool-call arguments, whose - * shape is model-authored and not known ahead of time. No other caller may walk - * untyped data: every message/content path is handled by a typed transformer. - */ -function mapJsonStrings(value: JsonValue, fn: (s: string) => string): JsonValue { - if (typeof value === "string") return fn(value); - if (Array.isArray(value)) { - let changed = false; - const out = value.map(item => { - const next = mapJsonStrings(item, fn); - if (next !== item) changed = true; - return next; - }); - return changed ? out : value; - } - if (value !== null && typeof value === "object") { - let changed = false; - const out: JsonRecord = {}; - for (const key of Object.keys(value)) { - const item = value[key]; - if (item === undefined) continue; - const next = mapJsonStrings(item, fn); - if (next !== item) changed = true; - out[key] = next; - } - return changed ? out : value; - } - return value; -} diff --git a/packages/coding-agent/src/secrets/placeholder-scan.ts b/packages/coding-agent/src/secrets/placeholder-scan.ts new file mode 100644 index 000000000..efed760fb --- /dev/null +++ b/packages/coding-agent/src/secrets/placeholder-scan.ts @@ -0,0 +1,506 @@ +import type { JsonRecord, JsonValue, SecretObfuscator } from "./obfuscator"; +import { + lookupFriendlyPlaceholderAlias, + PLACEHOLDER_RE, + type RegexScanSegment, + type ReplaceRegexScan, + resumePlaceholderScanAfterRejectedCandidate, +} from "./placeholder"; + +// ═══════════════════════════════════════════════════════════════════════════ +// Helpers +// ═══════════════════════════════════════════════════════════════════════════ + +// Like the untracked walk, but threads a parallel `origin` tag string through: +// preserved placeholder spans keep their existing origin tag (so a +// same-call-fresh "F" placeholder is never relabeled prior-call "I", and vice +// versa), while `transform`'s output — always freshly generated or redacted +// content in both callers below — is tagged "I" (it must not be re-matched as +// though it arrived in the input, mirroring plain-secret replacement tagging). +export function transformOutsidePlaceholdersTracked( + text: string, + origin: string, + shouldSkipPlaceholder: (placeholder: string) => boolean, + transform: (chunk: string) => string, + preservePlaceholder?: (placeholder: string) => string, +): { text: string; origin: string } { + PLACEHOLDER_RE.lastIndex = 0; + let result = ""; + let resultOrigin = ""; + let pendingIndex = 0; + for (;;) { + const match = PLACEHOLDER_RE.exec(text); + if (match === null) break; + if (!shouldSkipPlaceholder(match[0])) { + resumePlaceholderScanAfterRejectedCandidate(match); + continue; + } + const transformed = transform(text.slice(pendingIndex, match.index)); + result += transformed; + resultOrigin += "I".repeat(transformed.length); + const preserved = preservePlaceholder ? preservePlaceholder(match[0]) : match[0]; + result += preserved; + resultOrigin += origin.slice(match.index, match.index + match[0].length); + pendingIndex = match.index + match[0].length; + } + const trailing = transform(text.slice(pendingIndex)); + result += trailing; + resultOrigin += "I".repeat(trailing.length); + return { text: result, origin: resultOrigin }; +} + +export function trailingOutsidePreservedPlaceholderChunk( + text: string, + shouldPreservePlaceholder: (placeholder: string) => boolean, +): string { + PLACEHOLDER_RE.lastIndex = 0; + let pendingIndex = 0; + let sawPlaceholder = false; + for (;;) { + const match = PLACEHOLDER_RE.exec(text); + if (match === null) break; + if (!shouldPreservePlaceholder(match[0])) { + resumePlaceholderScanAfterRejectedCandidate(match); + continue; + } + sawPlaceholder = true; + pendingIndex = match.index + match[0].length; + } + return sawPlaceholder ? text.slice(pendingIndex) : ""; +} + +export function buildReplaceRegexScan( + text: string, + ranges: ReadonlyArray<{ start: number; end: number }>, + deobfuscateMap: ReadonlyMap, +): ReplaceRegexScan { + let scanText = ""; + let cursor = 0; + const segments: RegexScanSegment[] = []; + const appendSegment = ( + value: string, + textStart: number, + textEnd: number, + generatedPlaceholder: boolean, + recursive: boolean, + ) => { + if (value.length === 0) return; + const scanStart = scanText.length; + scanText += value; + segments.push({ + scanStart, + scanEnd: scanStart + value.length, + textStart, + textEnd, + generatedPlaceholder, + recursive, + }); + }; + + for (const range of ranges) { + appendSegment(text.slice(cursor, range.start), cursor, range.start, false, false); + const placeholder = text.slice(range.start, range.end); + const mapping = lookupFriendlyPlaceholderAlias(deobfuscateMap, placeholder); + appendSegment(mapping?.secret ?? placeholder, range.start, range.end, true, mapping?.recursive ?? false); + cursor = range.end; + } + appendSegment(text.slice(cursor), cursor, text.length, false, false); + + return { text: scanText, segments }; +} + +export function mapReplaceRegexMatch( + segments: ReadonlyArray, + scanStart: number, + scanEnd: number, +): { + start: number; + end: number; + recursive: boolean; + preserveGeneratedPlaceholders: boolean; + partialPlaceholderCut: boolean; + cutResumeIndex: number; + firstPlaceholderScanStart: number; +} { + const startSegment = findScanSegment(segments, scanStart); + const endSegment = findScanSegment(segments, scanEnd - 1); + const start = startSegment.generatedPlaceholder + ? startSegment.textStart + : startSegment.textStart + (scanStart - startSegment.scanStart); + const end = endSegment.generatedPlaceholder + ? endSegment.textEnd + : endSegment.textStart + (scanEnd - endSegment.scanStart); + // A match boundary that falls strictly inside a generated placeholder's + // expanded value cuts the underlying secret: the snap above pulls the span out + // to the whole `#…#` token, so the obfuscate path can leave it alone instead of + // consuming a partial placeholder expansion. + const partialPlaceholderCut = + (startSegment.generatedPlaceholder && scanStart > startSegment.scanStart) || + (endSegment.generatedPlaceholder && scanEnd < endSegment.scanEnd); + let recursive = false; + let preserveGeneratedPlaceholders = false; + // When the match straddles a placeholder, resume scanning just past the last + // overlapping placeholder so trailing wholly-outside content (e.g. an 8-char + // run after the secret) still gets matched instead of being consumed by the + // straddling span. `firstPlaceholderScanStart` marks where the leading + // wholly-outside prefix ends, so a prefix that independently matches can be + // redacted on its own rather than skipped along with the cut span. + let cutResumeIndex = scanStart; + let firstPlaceholderScanStart = -1; + for (const segment of segments) { + if (segment.scanStart >= scanEnd || segment.scanEnd <= scanStart) continue; + recursive ||= segment.recursive; + preserveGeneratedPlaceholders ||= segment.generatedPlaceholder; + if (segment.generatedPlaceholder) { + if (firstPlaceholderScanStart === -1) firstPlaceholderScanStart = segment.scanStart; + if (segment.scanEnd > cutResumeIndex) cutResumeIndex = segment.scanEnd; + } + } + return { + start, + end, + recursive, + preserveGeneratedPlaceholders, + partialPlaceholderCut, + cutResumeIndex, + firstPlaceholderScanStart, + }; +} + +function findScanSegment(segments: ReadonlyArray, scanIndex: number): RegexScanSegment { + for (const segment of segments) { + if (scanIndex >= segment.scanStart && scanIndex < segment.scanEnd) return segment; + } + throw new Error("regex match did not map to source text"); +} + +/** + * Extend a scan-space resume position past a consecutive run of generated + * placeholder segments starting exactly at it, with no raw gap in between. A + * cut-resolution resume point that happens to land precisely on the START of + * ANOTHER placeholder must not stop there and hand it to a fresh `regex.exec` + * attempt — the same content, scanned as an opaque adjacent placeholder run, + * must resolve identically whether the run's LEADING member is still raw text + * (this call is about to placeholder it) or is ALREADY a placeholder from a + * prior call or an earlier pass of this same call. Without this, a bounded + * regex whose reach spans two adjacent secrets plus trailing spillover bytes + * (e.g. `[A-Z]{9}` over `ABCDEFGH` + `SECRETUV` + `A`) resolves the leading + * secret as its own independent redaction on the FIRST obfuscate() call (a + * genuinely raw prefix gets its own match, then the discard for the rest + * resumes right after it), but on a LATER call — once that prefix is itself a + * placeholder — the very first match attempt starts already inside the + * placeholder run, cannot be prefix-narrowed at all, and its discard resume + * point lands mid-run instead of past it, exposing a shorter tail (`SECRETUV` + * + `A`) to a clean, un-cut match the first call never attempted. Chaining the + * resume point through every immediately-adjacent placeholder makes both + * calls land on the exact same next scan position. + */ +export function extendPastAdjacentPlaceholders(segments: ReadonlyArray, index: number): number { + let cursor = index; + for (;;) { + const segment = segments.find(candidate => candidate.scanStart === cursor && candidate.generatedPlaceholder); + if (!segment) return cursor; + cursor = segment.scanEnd; + } +} + +// Apply a fixed custom replacement across a matched span while preserving any +// inner generated placeholders. Usually the replacement is the user's single +// redaction marker for the whole match, so emit it for the first non-empty +// surrounding chunk and drop later chunks. But bounded regexes can cut through +// an already-emitted marker on the trailing side (`X#…#RED` from +// `XSECRETUVREDACTED`), where dropping the later prefix would leave raw bytes +// (`ACTED`) to be consumed on the next pass. Promote later chunks that are a +// prefix of the replacement to the FULL marker so the first pass is already a +// fixed point. The reversible placeholder stays intact in its relative +// position. +export function redactWithFixedReplacementOutsidePlaceholders( + text: string, + origin: string, + replacement: string, + shouldPreservePlaceholder: (placeholder: string) => boolean, +): { text: string; origin: string } { + let emitted = false; + return transformOutsidePlaceholdersTracked( + text, + origin, + shouldPreservePlaceholder, + chunk => { + if (chunk.length === 0) return ""; + if (!emitted) { + emitted = true; + return replacement; + } + return replacement.startsWith(chunk) ? replacement : ""; + }, + placeholder => placeholder, + ); +} + +export function deobfuscateGeneratedPlaceholderRanges( + text: string, + start: number, + end: number, + ranges: ReadonlyArray<{ start: number; end: number }>, + deobfuscateMap: ReadonlyMap, +): { text: string; recursive: boolean } { + let result = ""; + let cursor = start; + let recursive = false; + for (const range of ranges) { + if (range.end <= start || range.start >= end) continue; + const overlapStart = Math.max(range.start, start); + const overlapEnd = Math.min(range.end, end); + result += text.slice(cursor, overlapStart); + const placeholder = text.slice(overlapStart, overlapEnd); + const mapping = lookupFriendlyPlaceholderAlias(deobfuscateMap, placeholder); + result += mapping?.secret ?? placeholder; + recursive ||= mapping?.recursive ?? false; + cursor = overlapEnd; + } + result += text.slice(cursor, end); + return { text: result, recursive }; +} + +// Concatenate ONLY the deobfuscated placeholder ranges within [start, end), +// dropping the bytes that lie outside them. Used to test whether a regex match +// that straddles a prior-call placeholder would still match on the placeholder's +// own (expanded) secret value alone — i.e. the surrounding raw bytes are greedy +// spillover the match does not need, rather than content the match depends on. +export function placeholderInnerText( + text: string, + start: number, + end: number, + ranges: ReadonlyArray<{ start: number; end: number }>, + deobfuscateMap: ReadonlyMap, +): string { + let result = ""; + for (const range of ranges) { + if (range.end <= start || range.start >= end) continue; + const overlapStart = Math.max(range.start, start); + const overlapEnd = Math.min(range.end, end); + const placeholder = text.slice(overlapStart, overlapEnd); + const mapping = lookupFriendlyPlaceholderAlias(deobfuscateMap, placeholder); + result += mapping?.secret ?? placeholder; + } + return result; +} + +// Concatenate the bytes of [start, end) that lie OUTSIDE the given (ascending, +// non-overlapping) placeholder ranges. Used to test whether a regex match that +// straddles a prior-call placeholder would still match on its surrounding bytes +// alone — i.e. those bytes are genuinely new content to redact rather than a +// match that only exists because the deobfuscated placeholder bridges them. +export function textOutsidePlaceholderRanges( + text: string, + start: number, + end: number, + ranges: ReadonlyArray<{ start: number; end: number }>, +): string { + let result = ""; + let cursor = start; + for (const range of ranges) { + if (range.end <= start || range.start >= end) continue; + const overlapStart = Math.max(range.start, start); + const overlapEnd = Math.min(range.end, end); + result += text.slice(cursor, overlapStart); + cursor = overlapEnd; + } + result += text.slice(cursor, end); + return result; +} + +// Like `textOutsidePlaceholderRanges`, but tests each outside chunk against +// `regex` in its REAL context instead of on an isolated slice — tried in BOTH +// the literal `#…#` placeholder-token text AND the EXPANDED scan context +// (placeholder resolved to its secret value), since either can be the reason a +// chunk independently requires redaction: +// - Literal-token context matters when the placeholder TOKEN's own non-word +// boundary is what completes a boundary-sensitive pattern, e.g. a prefix +// "ABCDEFGH" next to a placeholder token matches `\b[A-Z]{8}\b` because the +// token's leading `#` is a non-word byte — but that boundary disappears +// once the placeholder expands into more `[A-Z]` bytes with no separator. +// - Expanded scan context matters when a lookbehind/lookahead only resolves +// once the neighboring placeholder is expanded, e.g. a prior plain +// placeholder for `ABCDEFGH` next to raw `SECRET`, matched by +// `(?<=ABCDEFGH)SECRET`: the literal placeholder token before `SECRET` +// never satisfies the lookbehind, so literal-context alone wrongly reports +// no independent match. +// A match only counts when it lies ENTIRELY within one outside chunk (in +// whichever context it was tested); a match that reaches into the +// placeholder itself is not evidence the outside chunk independently +// requires redaction. +export function outsidePlaceholderRangesAnyIndependentlyMatch( + text: string, + scanText: string, + segments: ReadonlyArray, + start: number, + end: number, + ranges: ReadonlyArray<{ start: number; end: number }>, + regex: RegExp, +): boolean { + // A text-space outside chunk lies entirely within one non-placeholder scan + // segment (placeholder ranges are exactly the gaps between such segments), + // so its scan-space span is a fixed offset from its text-space span. + const toScanSpace = (chunkStart: number, chunkEnd: number): [number, number] | undefined => { + for (const segment of segments) { + if (segment.generatedPlaceholder || segment.textStart > chunkStart || segment.textEnd < chunkEnd) continue; + const offset = segment.scanStart - segment.textStart; + return [chunkStart + offset, chunkEnd + offset]; + } + return undefined; + }; + const chunkIndependentlyMatches = (chunkStart: number, chunkEnd: number): boolean => { + if (chunkMatchesInSourceContext(text, chunkStart, chunkEnd, regex)) return true; + const scanSpan = toScanSpace(chunkStart, chunkEnd); + return scanSpan !== undefined && chunkMatchesInSourceContext(scanText, scanSpan[0], scanSpan[1], regex); + }; + let cursor = start; + for (const range of ranges) { + if (range.end <= start || range.start >= end) continue; + const overlapStart = Math.max(range.start, start); + const overlapEnd = Math.min(range.end, end); + if (cursor < overlapStart && chunkIndependentlyMatches(cursor, overlapStart)) return true; + cursor = overlapEnd; + } + return cursor < end && chunkIndependentlyMatches(cursor, end); +} + +// Whether `regex` (global) has a match fully contained in [chunkStart, chunkEnd) +// when run against the full `text` — so lookbehind/lookahead see the actual +// surrounding bytes rather than an isolated slice's edges. +function chunkMatchesInSourceContext(text: string, chunkStart: number, chunkEnd: number, regex: RegExp): boolean { + regex.lastIndex = chunkStart; + for (;;) { + const found = regex.exec(text); + if (found === null || found.index >= chunkEnd) return false; + const matchEnd = found.index + found[0].length; + if (matchEnd <= chunkEnd) return true; + regex.lastIndex = found[0].length === 0 ? found.index + 1 : matchEnd; + } +} + +export function firstOutsidePlaceholderRange( + start: number, + end: number, + ranges: ReadonlyArray<{ start: number; end: number }>, +): { start: number; end: number } | undefined { + let cursor = start; + for (const range of ranges) { + if (range.end <= start || range.start >= end) continue; + const overlapStart = Math.max(range.start, start); + const overlapEnd = Math.min(range.end, end); + if (cursor < overlapStart) return { start: cursor, end: overlapStart }; + cursor = overlapEnd; + } + return cursor < end ? { start: cursor, end } : undefined; +} + +export function countOutsidePlaceholderRanges( + start: number, + end: number, + ranges: ReadonlyArray<{ start: number; end: number }>, +): number { + let count = 0; + let cursor = start; + for (const range of ranges) { + if (range.end <= start || range.start >= end) continue; + const overlapStart = Math.max(range.start, start); + const overlapEnd = Math.min(range.end, end); + if (cursor < overlapStart) count++; + cursor = overlapEnd; + } + if (cursor < end) count++; + return count; +} + +export function replaceRange(text: string, start: number, end: number, replacement: string): string { + return text.slice(0, start) + replacement + text.slice(end); +} + +/** Deep-walk an object, transforming all string values. */ +export function deepWalkStrings(obj: T, transform: (s: string) => string): T { + if (typeof obj === "string") { + return transform(obj) as unknown as T; + } + if (Array.isArray(obj)) { + let changed = false; + const result = obj.map(item => { + const transformed = deepWalkStrings(item, transform); + if (transformed !== item) changed = true; + return transformed; + }); + return (changed ? result : obj) as unknown as T; + } + if (obj !== null && typeof obj === "object" && isPlainRecord(obj)) { + let changed = false; + const result: Record = {}; + for (const key of Object.keys(obj)) { + const value = (obj as Record)[key]; + const transformed = deepWalkStrings(value, transform); + if (transformed !== value) changed = true; + result[key] = transformed; + } + return (changed ? result : obj) as T; + } + return obj; +} + +function isPlainRecord(obj: object): obj is Record { + const prototype = Object.getPrototypeOf(obj); + return prototype === Object.prototype || prototype === null; +} + +export function collectJsonRegexSecretValues(obfuscator: SecretObfuscator, value: JsonValue): Set { + const values = new Set(); + const collect = (item: JsonValue): void => { + if (typeof item === "string") { + for (const secretValue of obfuscator.collectRegexSecretValuesForObfuscation(item)) { + values.add(secretValue); + } + return; + } + if (Array.isArray(item)) { + for (const child of item) collect(child); + return; + } + if (item !== null && typeof item === "object") { + for (const child of Object.values(item)) { + if (child !== undefined) collect(child); + } + } + }; + collect(value); + return values; +} + +/** + * Map every string in arbitrary JSON. Used ONLY for tool-call arguments, whose + * shape is model-authored and not known ahead of time. No other caller may walk + * untyped data: every message/content path is handled by a typed transformer. + */ +export function mapJsonStrings(value: JsonValue, fn: (s: string) => string): JsonValue { + if (typeof value === "string") return fn(value); + if (Array.isArray(value)) { + let changed = false; + const out = value.map(item => { + const next = mapJsonStrings(item, fn); + if (next !== item) changed = true; + return next; + }); + return changed ? out : value; + } + if (value !== null && typeof value === "object") { + let changed = false; + const out: JsonRecord = {}; + for (const key of Object.keys(value)) { + const item = value[key]; + if (item === undefined) continue; + const next = mapJsonStrings(item, fn); + if (next !== item) changed = true; + out[key] = next; + } + return changed ? out : value; + } + return value; +} diff --git a/packages/coding-agent/src/secrets/placeholder.ts b/packages/coding-agent/src/secrets/placeholder.ts new file mode 100644 index 000000000..7efd5776d --- /dev/null +++ b/packages/coding-agent/src/secrets/placeholder.ts @@ -0,0 +1,309 @@ +import * as crypto from "node:crypto"; +import type { SecretEntry } from "./obfuscator"; +import { ensureDistinctReplacement, generateDeterministicReplacement, REPLACEMENT_CHARS } from "./replacement"; + +// ═══════════════════════════════════════════════════════════════════════════ +// Placeholder format +// ═══════════════════════════════════════════════════════════════════════════ + +const HASH_CHARS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789"; +// Base length is sized for ~62 bits of entropy (64 bits of a keyed digest +// rendered as 12 base36 chars) so unrelated secrets do not collide on a shared +// base. A collision would let a persisted placeholder deobfuscate to the wrong +// secret when the configured secret set or its ordering changes across sessions. +const HASH_LEN = 12; +const MAX_FRIENDLY_NAME_LEN = 32; +// Plain/regex obfuscate matches shorter than this are toned down (never placed +// behind a reversible placeholder) to avoid redacting small words/fragments. +export const MIN_OBFUSCATE_SECRET_LEN = 8; + +// Per-process fallback key used when a caller does not supply a persisted +// per-install key. It is random (never shipped in source), so model-visible +// placeholders cannot be reversed by dictionary-hashing candidate secrets; it +// only forgoes cross-session token stability, which the persisted key provides. +let ephemeralPlaceholderKey: string | undefined; +export function defaultPlaceholderKey(): string { + ephemeralPlaceholderKey ??= crypto.randomBytes(32).toString("base64url"); + return ephemeralPlaceholderKey; +} + +type PlaceholderCaseHint = "U" | "L" | "C" | "M"; + +/** Normalize a friendly name into the model-visible placeholder prefix. */ +export function sanitizeSecretFriendlyName(name: string): string | undefined { + const sanitized = name + .replace(/[^A-Za-z0-9]/g, "") + .toUpperCase() + .slice(0, MAX_FRIENDLY_NAME_LEN); + return sanitized.length > 0 ? sanitized : undefined; +} + +/** + * Normalize a secret value into the same alnum-only, uppercased shape a + * friendly-name label or placeholder prefix is sanitized into, so comparing a + * raw (possibly lowercase/punctuated) secret value against already-sanitized, + * model-visible text does not miss a case- or separator-only variant. Unlike + * `sanitizeSecretFriendlyName` this never truncates and never signals "empty" + * via `undefined` — callers already guard on `.length > 0` before comparing. + */ +export function sanitizeForCollisionCheck(value: string): string { + return value.replace(/[^A-Za-z0-9]/g, "").toUpperCase(); +} + +// A label leaks a secret either by containing the whole normalized secret or, +// once it reaches the public display cap, by being the secret's visible prefix. +// Shorter names like "TOKEN" can still be intentional generic labels. +export function sanitizedLabelCollidesWithSecret(sanitizedLabel: string, sanitizedSecret: string): boolean { + if (sanitizedSecret.length === 0) return false; + if (sanitizedLabel.includes(sanitizedSecret)) return true; + return sanitizedLabel.length >= MAX_FRIENDLY_NAME_LEN && sanitizedSecret.startsWith(sanitizedLabel); +} + +/** + * Whether an entry needs the persisted placeholder key: either because it can + * produce a reversible (keyed) obfuscate-mode placeholder, or because a default + * (no custom `replacement`) replace-mode regex can reach + * `#generateRegexReplacement`'s key-derived idempotent fallback marker (see + * `#generateReplacement`) when every same-length candidate re-matches a + * pathological match-everything config (e.g. `[\s\S]{8}`). That fallback depends + * on the persisted per-install key — not just length — to stay a fixed point + * across a process restart; without a persisted key, a fresh install falls back + * to a process-random key (`defaultPlaceholderKey()`), so the fallback marker + * would churn across restarts even though the algorithm itself is stable. A + * regex WITH a custom `replacement` never reaches that fallback (it always emits + * the literal configured string), and a plain replace secret's replacement is + * pure content-hash (`#generateSecretReplacement`), so neither needs the key. + * Short plain obfuscate entries are toned down (never placeheld), so they must + * NOT force key creation: otherwise a `secret-placeholder.key` file is written + * and persisted for a config that ends up with no active secrets, leaving the + * key readable via a tool and reusable for later placeholders. + */ +export function secretEntryNeedsPlaceholderKey(entry: SecretEntry): boolean { + if ((entry.mode ?? "obfuscate") === "obfuscate") { + if (entry.type === "regex") return true; + return entry.content.length >= MIN_OBFUSCATE_SECRET_LEN; + } + return entry.type === "regex" && entry.replacement === undefined; +} + +/** + * Whether a plain replace-mode replacement string can contribute a fragment that + * helps the replace phase reconstruct an obfuscate `content`. During obfuscate()'s + * replace phase the output is a tiling of passthrough bytes (adversary-controlled + * provider text) and whole replacement outputs; any contiguous occurrence of + * `content` in that output is covered by interior replacement tiles (each a + * substring of `content`) bordered by passthrough at the ends, where the border + * tile may be a suffix of a replacement (forming `content`'s prefix) or a prefix + * of a replacement (forming `content`'s suffix). An EMPTY replacement deletes its + * trigger entirely, joining the passthrough on both sides; with adversary-chosen + * surrounding bytes that can form any non-empty `content` across the deleted gap. + * So a replacement can help iff it is empty, is a substring of `content`, + * contains `content`, or shares such a border overlap. + */ +function replacementCanFormContent(replacement: string, content: string): boolean { + if (replacement.length === 0) return content.length > 0; + if (content.includes(replacement) || replacement.includes(content)) return true; + const maxOverlap = Math.min(replacement.length, content.length); + for (let k = 1; k <= maxOverlap; k++) { + // A suffix of the replacement forms the prefix of the content (left border), + // or a prefix of the replacement forms the suffix of the content (right border). + if (content.startsWith(replacement.slice(replacement.length - k)) || content.endsWith(replacement.slice(0, k))) { + return true; + } + } + return false; +} + +/** + * Whether a SET of entries needs the persisted placeholder key. `obfuscate()` + * applies plain replace-mode mappings before the plain-obfuscate pass, so a plain + * obfuscate entry only emits a reversible (keyed) placeholder when its content can + * still appear AFTER the replace phase. When no obfuscate entry can ever produce a + * placeholder, the persisted key must NOT be required/created — otherwise an + * effectively replace-only secret set still writes `secret-placeholder.key` and + * fails startup when the agent config dir is unwritable. + * + * The decision models the replace phase as the obfuscator actually runs it: + * replace mappings are content-keyed (later duplicate wins) and applied in + * descending content-length order; for a fresh probe (no prior placeholders) that + * phase is plain sequential substring replacement. A plain obfuscate entry needs + * the key when its content survives that simulated phase (direct typing) OR when + * any effective replacement can form the content via tiling — a substring, + * wholesale superstring, or prefix/suffix border that joins with surrounding + * passthrough bytes (see `replacementCanFormContent`). This covers direct + * shadowing (`SECRET -> safe`), reintroduction, duplicate ordering, transitive + * chains, and context-joined fragments uniformly. Default (omitted) replacements + * are deterministic, length-preserving, and distinct, so a same-content shadow + * with no other interacting replacement stays key-free. + * Replacement outputs are themselves rewritten by every later (shorter-content) + * replacement before the plain-obfuscate pass sees them, so a fragment that a + * subsequent replacement erases (`AA -> SEC` then `S -> X` turns every `SEC` into + * `XEC`) no longer forces the key. Surrounding bytes stay modeled as arbitrary + * passthrough, so testing the surviving fragment only drops false positives and + * never under-approximates a real key need. + */ +export function secretEntriesNeedPlaceholderKey(entries: SecretEntry[]): boolean { + const replaceMap = new Map(); + for (const entry of entries) { + if (entry.type !== "plain" || (entry.mode ?? "obfuscate") !== "replace") continue; + replaceMap.set( + entry.content, + entry.replacement ?? ensureDistinctReplacement(generateDeterministicReplacement(entry.content), entry.content), + ); + } + const replacePhase = [...replaceMap].sort((a, b) => b[0].length - a[0].length); + // Apply the replace phase from `start` onward. The phase runs in descending + // content-length order, so a replacement output emitted at index i is rewritten + // only by the later (shorter-content) replacements at i+1…; `start` 0 models a + // value typed directly into the input. + const applyReplacePhaseFrom = (text: string, start: number): string => { + let result = text; + for (let i = start; i < replacePhase.length; i++) { + result = result.split(replacePhase[i][0]).join(replacePhase[i][1]); + } + return result; + }; + return entries.some(entry => { + if (!secretEntryNeedsPlaceholderKey(entry)) return false; + // Regex obfuscate entries match dynamically; conservatively require the key. + if (entry.type !== "plain") return true; + const content = entry.content; + if (applyReplacePhaseFrom(content, 0).includes(content)) return true; + // Test each replacement output in the form it SURVIVES the rest of the phase, + // so a fragment a later replacement erases no longer forces the key. The + // content it tiles into must also survive those later replacements: if a + // shorter-content replacement rewrites the surrounding passthrough bytes + // (e.g. `AA -> SEC` forms `SEC`+`RET12`, then `R -> X` turns the freshly + // formed `SECRET12` into `SECXET12`), the content can never reach the + // obfuscate pass, so the key is not needed. Requiring content stability only + // drops such false positives — a formation that genuinely survives is still + // caught at the replacement index that produces it. + return replacePhase.some( + ([, replacement], i) => + applyReplacePhaseFrom(content, i + 1) === content && + replacementCanFormContent(applyReplacePhaseFrom(replacement, i + 1), content), + ); + }); +} + +// Derive the model-visible base from a KEYED digest of the secret. xxHash is +// fast and unkeyed, so a fixed-seed content hash of a low-entropy secret could +// be dictionaried from the transcript; HMAC-SHA256 under a private per-install +// key cannot, since the attacker lacks the key. +export function buildHashBase(key: string, value: string): string { + const digest = new Bun.CryptoHasher("sha256", key).update(value).digest(); + let v = 0n; + for (let i = 0; i < 8; i++) v = (v << 8n) | BigInt(digest[i]); + const radix = BigInt(HASH_CHARS.length); + let tag = ""; + for (let i = 0; i < HASH_LEN; i++) { + tag += HASH_CHARS[Number(v % radix)]; + v /= radix; + } + return tag; +} + +// Build a deterministic, key-derived run of REPLACEMENT_CHARS of the given +// length. Used to redact a per-chunk replace remainder to a marker that depends +// only on the per-install key and the remainder length, so a fresh obfuscator +// reproduces the identical marker (idempotent redaction across restarts) while +// the run stays unpredictable without the key (raw sentinel-shaped bytes cannot +// equal the marker, so they are still redacted rather than passed through). +export function buildKeyedReplacementRun(key: string, length: number): string { + if (length <= 0) return ""; + const radix = REPLACEMENT_CHARS.length; + let out = ""; + for (let block = 0; out.length < length; block++) { + const digest = new Bun.CryptoHasher("sha256", key).update(`replace-chunk\0${length}\0${block}`).digest(); + for (let i = 0; i < digest.length && out.length < length; i++) { + out += REPLACEMENT_CHARS[digest[i] % radix]; + } + } + return out; +} + +export function inferCaseHint(secret: string): PlaceholderCaseHint | undefined { + let hasCased = false; + let hasUpper = false; + let hasLower = false; + let capitalized = true; + let seenFirstCased = false; + + for (let i = 0; i < secret.length; i++) { + const code = secret.charCodeAt(i); + const isUpper = code >= 65 && code <= 90; + const isLower = code >= 97 && code <= 122; + if (!isUpper && !isLower) continue; + + hasCased = true; + if (isUpper) { + hasUpper = true; + if (seenFirstCased) capitalized = false; + } else { + hasLower = true; + if (!seenFirstCased) capitalized = false; + } + seenFirstCased = true; + } + + if (!hasCased) return undefined; + if (hasUpper && !hasLower) return "U"; + if (hasLower && !hasUpper) return "L"; + if (capitalized) return "C"; + return "M"; +} + +export function buildPlaceholder(hint: PlaceholderCaseHint | undefined, base: string, friendlyName?: string): string { + const prefix = friendlyName ? `${friendlyName}_` : ""; + return hint ? `$$${prefix}${base}:${hint}$$` : `$$${prefix}${base}$$`; +} + +/** Regex matching `$$HASH$$`, `$$HASH:U$$`, and `$$FRIENDLY_HASH(:hint)$$` placeholders. */ +export const PLACEHOLDER_RE = /\$\$(?:[A-Z0-9]+_)?[A-Z0-9]{4,}(?::[ULCM])?\$\$/g; + +export function resumePlaceholderScanAfterRejectedCandidate(match: RegExpExecArray): void { + // RegExp#exec does not find overlapping matches. Restart at the rejected + // candidate's closing delimiter, which can open an immediately adjacent placeholder. + PLACEHOLDER_RE.lastIndex = match.index + match[0].length - 2; +} + +export function placeholderWithoutFriendlyName(placeholder: string): string | undefined { + const match = /^\$\$[A-Z0-9]+_([A-Z0-9]{4,}(?::[ULCM])?)\$\$$/.exec(placeholder); + return match ? `$$${match[1]}$$` : undefined; +} + +export function lookupFriendlyPlaceholderAlias( + deobfuscateMap: ReadonlyMap, + placeholder: string, +): { secret: string; recursive: boolean } | undefined { + const direct = deobfuscateMap.get(placeholder); + if (direct !== undefined) return direct; + const unprefixed = placeholderWithoutFriendlyName(placeholder); + return unprefixed !== undefined ? deobfuscateMap.get(unprefixed) : undefined; +} + +const PENDING_PLACEHOLDER_SUFFIX_RE = /(?:\$\$(?:[A-Z0-9]+_)?[A-Z0-9]*(?::[ULCM]?)?|\$)$/; + +// Withhold a trailing run that could be the start of a placeholder from streamed +// deltas, so a partial token is never emitted before deobfuscation can replace +// it. A lone trailing delimiter character is always buffered because it can open +// a placeholder; the final non-streamed flush re-emits it when no token follows. +export function stripPendingSecretPlaceholderSuffix(text: string): string { + const pendingPlaceholderStart = text.match(PENDING_PLACEHOLDER_SUFFIX_RE); + if (pendingPlaceholderStart?.index === undefined) return text; + return text.slice(0, pendingPlaceholderStart.index); +} + +export interface RegexScanSegment { + scanStart: number; + scanEnd: number; + textStart: number; + textEnd: number; + generatedPlaceholder: boolean; + recursive: boolean; +} + +export interface ReplaceRegexScan { + text: string; + segments: RegexScanSegment[]; +} diff --git a/packages/coding-agent/src/secrets/replacement.ts b/packages/coding-agent/src/secrets/replacement.ts new file mode 100644 index 000000000..05b9a1060 --- /dev/null +++ b/packages/coding-agent/src/secrets/replacement.ts @@ -0,0 +1,216 @@ +// ═══════════════════════════════════════════════════════════════════════════ +// Deterministic replacement generation +// ═══════════════════════════════════════════════════════════════════════════ + +export const REPLACEMENT_CHARS = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789"; +const NONMATCHING_REPLACEMENT_CHARS = `${REPLACEMENT_CHARS}!#$%&()*+,-./:;<=>?@[]^_{|}~`; +// Whitespace bytes used to build last-resort redactions for a default replace +// regex that matches every non-whitespace candidate (e.g. `\S{n}`). Only +// `space`/`tab` are used — never a line terminator — so a `.`-style +// match-everything regex (which matches space and tab but not `\n`) still +// exhausts to the sentinel instead of redacting to a newline run. +const WHITESPACE_REPLACEMENT_CHARS = " \t"; + +/** Generate a deterministic same-length replacement string from a secret value. */ +export function generateDeterministicReplacement(secret: string): string { + if (secret.length === 0) return ""; + // Prefix generated chunks with a fixed `ZZ` so re-redacting an already-emitted + // 1–2 char chunk is a fixed point (the deterministic replacement of a <=2-char + // value is itself `Z`/`ZZ`), keeping short default-replacement remainders next + // to a reversible placeholder stable across an obfuscator restart. + const hash = BigInt(Bun.hash(secret)); + const chars = secret.length === 1 ? ["Z"] : ["Z", "Z"]; + let h = hash; + for (let i = chars.length; i < secret.length; i++) { + h = h ^ (BigInt(i + 1) * 0x9e3779b97f4a7c15n); + const idx = Number((h < 0n ? -h : h) % BigInt(REPLACEMENT_CHARS.length)); + chars.push(REPLACEMENT_CHARS[idx]); + } + return chars.join(""); +} + +/** + * Force a length-preserving deterministic replacement to differ from the secret + * it stands in for. `generateDeterministicReplacement` seeds its first 1–2 chars + * with the `Z`/`ZZ` sentinel, so a whole configured value that is exactly `Z` or + * `ZZ` (or an astronomically unlikely longer hash collision) would otherwise be + * emitted unchanged and ship the raw secret to the provider. Flip the first char + * to a fixed different glyph: same length, still deterministic, guaranteed != the + * secret. Only safe for a whole CONFIGURED value (a plain secret matches its own + * literal, so the perturbed output is no longer matched and stays a fixed point); + * per-chunk remainders must keep the sentinel to remain idempotent across restart. + */ +export function ensureDistinctReplacement(replacement: string, secret: string): string { + if (replacement.length === 0 || replacement !== secret) return replacement; + const alt = replacement[0] === REPLACEMENT_CHARS[0] ? REPLACEMENT_CHARS[1] : REPLACEMENT_CHARS[0]; + return alt + replacement.slice(1); +} + +// How far left of the matched span the re-match scan begins looking for a match +// that overlaps the candidate. This bounds ONLY the match-start search position, +// never the lookbehind/lookahead context: the probe below substitutes the +// candidate into the FULL text, so a regex's lookbehind/lookahead assertions +// always evaluate against complete context regardless of width. The single +// re-match this misses is one that begins more than this many bytes before the +// span and extends into it (a single match longer than the window) — that only +// churns the chosen redaction marker between candidates, never back to the raw +// matched value, so it cannot leak a secret. +const REGEX_REMATCH_BACKSCAN = 512; + +export interface RegexMatchContext { + /** Full text the match was found in (positions are offsets into it). */ + text: string; + /** Start/end of the matched span being replaced. */ + start: number; + end: number; +} + +/** + * Whether `candidate`, substituted for the matched span in its surrounding text, + * is re-matched by `regex` at its own position. A replace-mode regex that depends + * on context (lookbehind/lookahead/`\b`) can match a candidate that does NOT match + * in isolation: e.g. `(?<=api=)[AZ]` never matches a bare `A`, but `api=A` does, so + * a candidate `A` chosen by an isolation test is re-redacted on the next obfuscate() + * pass and can oscillate back to the raw matched value. The probe substitutes the + * candidate into the FULL text — not a truncated window — so a wide lookbehind or + * lookahead (e.g. `(?<=A{600})`) still evaluates against the context that makes it + * match. Truncating that context dropped the assertion's reach and falsely + * accepted an oscillating, leaky candidate. The scan starts a bounded distance + * left of the span and stops once a match begins at/after the span's end (matches + * arrive in order), keeping per-candidate cost independent of total text length. + */ +export function regexRematchesInContext(candidate: string, regex: RegExp, ctx: RegexMatchContext): boolean { + const probe = ctx.text.slice(0, ctx.start) + candidate + ctx.text.slice(ctx.end); + const spanStart = ctx.start; + const spanEnd = spanStart + candidate.length; + regex.lastIndex = Math.max(0, spanStart - REGEX_REMATCH_BACKSCAN); + for (let m = regex.exec(probe); m !== null; m = regex.exec(probe)) { + const matchStart = m.index; + const matchEnd = m.index + m[0].length; + // Matches arrive in increasing position; once one starts at or past the + // span's end it cannot cover the candidate, and neither can any later one. + if (matchStart >= spanEnd) break; + // A match overlapping the candidate's own bytes means those bytes get + // re-redacted on a later pass — not a fixed point. + if (matchEnd > spanStart) return true; + // Zero-width matches do not advance lastIndex; step past to avoid a loop. + if (m[0].length === 0) regex.lastIndex++; + } + return false; +} + +/** + * Search same-length replacements for one the regex does NOT match, so a default + * regex secret whose deterministic replacement collides with its own value (the + * `Z`/`ZZ` sentinel, or an astronomical hash collision) is still redacted to a + * STABLE nonmatching value instead of shipping the raw secret. A nonmatching + * candidate is a fixed point under re-obfuscation — the regex never re-matches it, + * so it cannot re-leak on a later pass. The search stays bounded to O(length * + * alphabet) regardless of value length: first exhaust every single-position + * substitution against a deterministic baseline (`AAAA…`, then `!AAA…`, `A!AA…`, + * …) so any regex that only needs one out-of-class byte — regardless of position — + * is found in a handful of probes rather than enumerating every combination (which + * for a 3-byte match-everything config, e.g. `[\s\S]{3}`, would otherwise run + * 90**3 = 729000 candidates through the regex on every single match, stalling + * provider requests). Candidates are enumerated deterministically over a stable + * ASCII alphabet: alphanumerics first (usually enough), then punctuation fallback + * bytes when the regex covers every alphanumeric candidate. When the regex still + * matches around a lone perturbed byte (for example `[A-Za-z0-9].*` matching the + * unperturbed tail), full-width same-byte candidates (`!!!!!`, `_____`, …) are + * tried next. When the regex covers every non-whitespace candidate (e.g. `\S{n}`), + * whitespace markers (a full space/tab run, then a single whitespace byte among + * non-whitespace filler) are tried as a last resort. A genuine match-everything + * regex (`.`/`[\s\S]`, which also matches space and tab) still exhausts this bounded + * sweep and returns undefined, letting the caller keep its own fixed-point fallback + * — bounded search can in principle miss an escape that depends jointly on + * multiple positions in a way no single-position swap reaches, but no realistic + * secret-redaction regex (character classes, literal matches, anchored/bounded + * repeats) has that shape. + */ +export function findNonMatchingReplacement( + value: string, + regex: RegExp, + context: RegexMatchContext, +): string | undefined { + const len = value.length; + if (len === 0) return undefined; + // Exhaust every single-position substitution against the deterministic baseline + // first (covers the common case cheaply), then fall back to full-width same-byte + // candidates for a regex that only rejects a lone perturbed byte in context. + const baseline = NONMATCHING_REPLACEMENT_CHARS[0].repeat(len); + for (let position = 0; position < len; position++) { + for (const ch of NONMATCHING_REPLACEMENT_CHARS) { + const candidate = `${baseline.slice(0, position)}${ch}${baseline.slice(position + 1)}`; + if (candidate === value) continue; + if (!regexRematchesInContext(candidate, regex, context)) return candidate; + } + } + // If the regex can still match around a lone punctuation byte (for example + // `[A-Za-z0-9].*` matching the `AAAA` tail of `!AAAA`), try full-width + // same-byte fallbacks like `!!!!!`, `_____`, etc. before giving up. + for (const ch of NONMATCHING_REPLACEMENT_CHARS) { + const candidate = ch.repeat(len); + if (candidate === value) continue; + if (!regexRematchesInContext(candidate, regex, context)) return candidate; + } + return findWhitespaceFallbackReplacement(value, regex, context); +} + +/** + * Last-resort fallback for a default replace regex that matches every + * non-whitespace candidate. Builds same-length whitespace markers the regex + * cannot match: first a full space/tab run (handles `\S`-class patterns), then a + * single whitespace byte among non-whitespace filler (` AAAA`, `A AAA`, …). The + * mixed marker defeats regexes that ALSO match all-space/all-tab runs, e.g. + * `(?:\S{n}| {n}|\t{n})`, because the lone whitespace byte breaks every + * fixed-length run. A genuine match-everything regex (`.`/`[\s\S]`) matches the + * filler and the whitespace alike, so this still returns undefined there, keeping + * the caller's sentinel as the sole fixed point. + */ +function findWhitespaceFallbackReplacement( + value: string, + regex: RegExp, + context: RegexMatchContext, +): string | undefined { + const len = value.length; + const filler = NONMATCHING_REPLACEMENT_CHARS[0]; + for (const ws of WHITESPACE_REPLACEMENT_CHARS) { + const full = ws.repeat(len); + if (full !== value) { + if (!regexRematchesInContext(full, regex, context)) return full; + } + for (let pos = 0; pos < len; pos++) { + const candidate = `${filler.repeat(pos)}${ws}${filler.repeat(len - pos - 1)}`; + if (candidate === value) continue; + if (!regexRematchesInContext(candidate, regex, context)) return candidate; + } + } + return undefined; +} + +/** + * Whether a default (no custom `replacement`) replace-mode regex can never + * safely redact a 1-2 char match: `findNonMatchingReplacement`'s bounded + * search — the same search `#generateRegexReplacement` runs at match time — + * finds no candidate the regex fails to re-match. This holds independent of + * any actual per-install key: the search already exhausts every character in + * `REPLACEMENT_CHARS` (the alphabet `buildKeyedReplacementRun` draws its + * fallback marker from) plus punctuation and whitespace, so if none of those + * escape the regex, no key-derived marker drawn from the same alphabet can + * either — the marker is guaranteed to re-match too, making every such match + * unresolvable: the fallback could only ever emit the raw matched text + * unchanged. Probed with a value (`"\0".repeat(length)`) the bounded search + * never treats as a real candidate, so the result depends only on the + * regex's own matching behavior, not on this specific probe. + */ +export function regexHasUnresolvableShortMatchFallback(regex: RegExp): boolean { + return ([1, 2] as const).some(length => { + const probe = "\u0000".repeat(length); + const savedLastIndex = regex.lastIndex; + try { + return findNonMatchingReplacement(probe, regex, { text: probe, start: 0, end: length }) === undefined; + } finally { + regex.lastIndex = savedLastIndex; + } + }); +} diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 707e87a5b..a2ae42c32 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -174,8 +174,8 @@ import { deobfuscateSessionContext, deobfuscateToolArguments, obfuscateProviderContext, - type SecretObfuscator, -} from "../secrets/obfuscator"; +} from "../secrets/message-transform"; +import type { SecretObfuscator } from "../secrets/obfuscator"; import { AUTO_THINKING, type ConfiguredThinkingLevel, diff --git a/packages/coding-agent/src/session/session-handoff.ts b/packages/coding-agent/src/session/session-handoff.ts index 467e3d333..eef73ff39 100644 --- a/packages/coding-agent/src/session/session-handoff.ts +++ b/packages/coding-agent/src/session/session-handoff.ts @@ -14,7 +14,8 @@ import { logger, Snowflake } from "@oh-my-pi/pi-utils"; import type { ModelRegistry } from "../config/model-registry"; import type { Settings } from "../config/settings"; import type { ExtensionRunner, SessionBeforeSwitchResult } from "../extensibility/extensions"; -import { obfuscateProviderContext, type SecretObfuscator } from "../secrets/obfuscator"; +import { obfuscateProviderContext } from "../secrets/message-transform"; +import type { SecretObfuscator } from "../secrets/obfuscator"; import type { HandoffResult, SessionHandoffOptions } from "./agent-session-types"; import type { BashSessionTransition } from "./bash-runner"; import type { SessionContext } from "./session-context"; diff --git a/packages/coding-agent/src/session/session-provider-boundary.ts b/packages/coding-agent/src/session/session-provider-boundary.ts index 28191e3ef..66754cd89 100644 --- a/packages/coding-agent/src/session/session-provider-boundary.ts +++ b/packages/coding-agent/src/session/session-provider-boundary.ts @@ -10,12 +10,9 @@ import { formatModelString } from "../config/model-resolver"; import type { Settings } from "../config/settings"; import { validateProviderMaxInFlightRequests } from "../config/settings"; import type { LocalProtocolOptions } from "../internal-urls"; -import { - deobfuscateSessionContext, - obfuscateMessages, - type SecretObfuscator, - stripPendingSecretPlaceholderSuffix, -} from "../secrets/obfuscator"; +import { deobfuscateSessionContext, obfuscateMessages } from "../secrets/message-transform"; +import type { SecretObfuscator } from "../secrets/obfuscator"; +import { stripPendingSecretPlaceholderSuffix } from "../secrets/placeholder"; import { normalizeModelContextImages } from "../utils/image-loading"; import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback"; import { type CustomMessage, convertToLlm } from "./messages"; diff --git a/packages/coding-agent/test/secrets-obfuscator.test.ts b/packages/coding-agent/test/secrets-obfuscator.test.ts index 3b1927daf..f4879028c 100644 --- a/packages/coding-agent/test/secrets-obfuscator.test.ts +++ b/packages/coding-agent/test/secrets-obfuscator.test.ts @@ -23,13 +23,14 @@ import { obfuscateMessages, obfuscateProviderContext, obfuscateToolArguments, - type SecretEntry, - SecretObfuscator, +} from "@oh-my-pi/pi-coding-agent/secrets/message-transform"; +import { type SecretEntry, SecretObfuscator } from "@oh-my-pi/pi-coding-agent/secrets/obfuscator"; +import { sanitizeSecretFriendlyName, secretEntriesNeedPlaceholderKey, secretEntryNeedsPlaceholderKey, stripPendingSecretPlaceholderSuffix, -} from "@oh-my-pi/pi-coding-agent/secrets/obfuscator"; +} from "@oh-my-pi/pi-coding-agent/secrets/placeholder"; import { compileSecretRegex } from "@oh-my-pi/pi-coding-agent/secrets/regex"; import { getActiveProfile, getAgentDir, setProfile } from "@oh-my-pi/pi-utils/dirs";