diff --git a/AGENTS.md b/AGENTS.md index 1630cc37a..84cbf45e1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,6 +29,7 @@ This repo contains multiple packages, but **`packages/coding-agent/`** is the pr - **Class privacy**: use ES `#private` fields; leave externally accessible members bare. **No `private`/`protected`/`public` keyword on fields or methods**, except on **constructor parameter properties** where TypeScript requires it (e.g. `constructor(private readonly session: ToolSession)`). - **Promises**: use `Promise.withResolvers()` instead of `new Promise((resolve, reject) => ...)`. - **Prompts**: never build prompts in code (no inline strings, template literals, or concatenation). Prompts live in static `.md` files; use Handlebars for dynamic content. Import them via `import content from "./prompt.md" with { type: "text" }` — not `readFile`. +- **Worker scripts**: reference the entry via `import workerUrl from "./worker.ts" with { type: "file" }` so the path survives bundling. tsgo flags it (TS1192/TS5097) — suppress with `// @ts-expect-error -- Bun file-URL import` directly above the line (see `tab-supervisor.ts`, `context-manager.ts`, `aggregator.ts`). ## Bun Over Node diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ee6e211f7..9b4546a8f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,12 +1,14 @@ # Changelog ## [Unreleased] + ### Breaking Changes - Changed the `timeoutMs` execution option to no longer be enforced during worker-based JS runs, so callers must rely on external cancellation signals for time limits ### Added +- Added a live single-line sync progress display to the stats command showing current/total sessions while syncing - Added automatic inline JS evaluation fallback when worker creation failed so script execution still works in environments without worker support ### Changed diff --git a/packages/coding-agent/src/cli/stats-cli.ts b/packages/coding-agent/src/cli/stats-cli.ts index 899ed8db8..f6a29e887 100644 --- a/packages/coding-agent/src/cli/stats-cli.ts +++ b/packages/coding-agent/src/cli/stats-cli.ts @@ -8,6 +8,58 @@ import { APP_NAME, formatDuration, formatNumber, formatPercent } from "@oh-my-pi import chalk from "chalk"; import { openPath } from "../utils/open"; +/** + * Single-line TTY progress bar. On a non-TTY stream we just stay quiet - + * the final "Synced ..." summary still prints either way. + */ +function createSyncProgressReporter(): { + onProgress: (event: { current: number; total: number; sessionFile: string }) => void; + finish: () => void; +} { + const stream = process.stderr; + const isTty = stream.isTTY === true; + let lastWidth = 0; + let lastRender = 0; + return { + onProgress(event) { + if (!isTty) return; + const now = Date.now(); + // Throttle to ~30 fps and always force a render for the last file. + if (event.current < event.total && now - lastRender < 33) return; + lastRender = now; + const label = chalk.dim(shortenSessionFile(event.sessionFile)); + const pct = ((event.current / event.total) * 100).toFixed(0).padStart(3, " "); + const counter = chalk.cyan(`[${event.current}/${event.total}]`); + const line = `${counter} ${pct}% ${label}`; + const columns = stream.columns ?? 120; + const trimmed = truncateToColumns(line, columns - 1); + stream.write(`\r${trimmed.padEnd(lastWidth)}`); + lastWidth = trimmed.length; + }, + finish() { + if (!isTty || lastWidth === 0) return; + stream.write(`\r${" ".repeat(lastWidth)}\r`); + lastWidth = 0; + }, + }; +} + +function shortenSessionFile(p: string): string { + const marker = "/sessions/"; + const idx = p.indexOf(marker); + return idx >= 0 ? p.slice(idx + marker.length) : p; +} + +function truncateToColumns(s: string, max: number): string { + if (max <= 0) return ""; + const width = Bun.stringWidth(s, { countAnsiEscapeCodes: false }); + if (width <= max) return s; + // Cheap right-trim with an ellipsis - we don't need ANSI-aware slicing + // because the colored prefix is short and the truncated tail is the + // dim filename, where dropping bytes is fine. + return `${s.slice(0, Math.max(0, max - 1))}\u2026`; +} + // ============================================================================= // Types // ============================================================================= @@ -74,8 +126,10 @@ export async function runStatsCommand(cmd: StatsCommandArgs): Promise { ); // Sync session files first - console.log("Syncing session files..."); - const { processed, files } = await syncAllSessions(); + const progress = createSyncProgressReporter(); + process.stderr.write("Syncing session files...\n"); + const { processed, files } = await syncAllSessions({ onProgress: progress.onProgress }); + progress.finish(); const total = await getTotalMessageCount(); console.log(`Synced ${processed} new entries from ${files} files (${total} total)\n`); diff --git a/packages/coding-agent/src/eval/js/worker-core.ts b/packages/coding-agent/src/eval/js/worker-core.ts index fca1139b5..78dcb4412 100644 --- a/packages/coding-agent/src/eval/js/worker-core.ts +++ b/packages/coding-agent/src/eval/js/worker-core.ts @@ -503,6 +503,7 @@ function indirectEval(source: string, filename?: string): unknown { const withPragma = filename ? `${source}\n//# sourceURL=${filename}` : source; // Read `eval` via a property access so the call site is *indirect* (global scope), // not direct (this function's lexical scope). The cast erases the lib.dom return type. + // biome-ignore lint/security/noGlobalEval: indirect eval is intentional for sandbox execution const geval = globalThis.eval as (src: string) => unknown; return geval(withPragma); } diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index 9e5dfa6f5..e82f2d38e 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -1,6 +1,26 @@ # Changelog ## [Unreleased] +### Breaking Changes + +- Broke backward compatibility of behavior stats fields by replacing `yellingSentences`/`dramaRuns` with `yelling`/`anguish` and adding `negation`, `repetition`, `blame` in query result types and persisted `user_messages` schema + +### Added + +- Added `SyncOptions` to `syncAllSessions` with `onProgress` and `workers` to optionally show per-file sync progress and tune parser concurrency +- Added new frustration behavior metrics (`negation`, `repetition`, `blame`) plus a `frustration` aggregate in behavior charts, model tables, and summary cards + +### Changed + +- Changed sync ingestion to parse session files through a worker pool while applying parsed results and database writes on the main thread +- Changed behavior analysis to strip code blocks, XML/URLs, quoted lines, and placeholders before scoring and to suppress signals on long structured messages +- Changed dashboard metrics labels and totals to the new signal names, including replacing the old three-signal totals with `yelling`, `profanity`, `anguish`, and `frustration` +- Changed sync output to print a live terminal progress indicator while processing session files + +### Fixed + +- Fixed user-message attribution so assistant model/provider links are backfilled during incremental sync instead of being left unknown +- Fixed word-boundary regex handling in profanity detection so matching now works as intended in normal prose ## [14.9.5] - 2026-05-12 diff --git a/packages/stats/src/aggregator.ts b/packages/stats/src/aggregator.ts index dda0a3857..c243564f1 100644 --- a/packages/stats/src/aggregator.ts +++ b/packages/stats/src/aggregator.ts @@ -19,67 +19,177 @@ import { insertMessageStats, insertUserMessageStats, setFileOffset, + updateUserMessageLinks, } from "./db"; -import { getSessionEntry, listAllSessionFiles, parseSessionFile } from "./parser"; +import { getSessionEntry, listAllSessionFiles, type ParseSessionResult } from "./parser"; +import type { SyncWorkerRequest, SyncWorkerResponse } from "./sync-worker"; +// `with { type: "file" }` resolves to the worker's absolute path at runtime +// (dev) and survives bundling (the asset is copied alongside the build). +// tsgo doesn't recognize Bun's file-URL import attribute and would raise +// TS1192/TS5097 here; Bun honors it. Same suppression pattern lives in +// `tab-supervisor.ts` and `context-manager.ts`. +// @ts-expect-error -- Bun file-URL import (see comment above). +import syncWorkerUrl from "./sync-worker.ts" with { type: "file" }; import type { BehaviorDashboardStats, DashboardStats, MessageStats, RequestDetails } from "./types"; /** - * Sync a single session file to the database. - * Only processes new entries since the last sync. + * Apply a freshly parsed result to the database. Runs entirely on the + * main thread so the single SQLite handle owns every write. */ -async function syncSessionFile(sessionFile: string): Promise { - // Get file stats - let fileStats: fs.Stats; - try { - fileStats = await fs.promises.stat(sessionFile); - } catch { - return 0; +function applyParseResult(sessionFile: string, lastModified: number, result: ParseSessionResult): number { + if (result.stats.length > 0) insertMessageStats(result.stats); + if (result.userStats.length > 0) insertUserMessageStats(result.userStats); + if (result.userLinks.length > 0) updateUserMessageLinks(result.userLinks); + setFileOffset(sessionFile, result.newOffset, lastModified); + return result.stats.length + result.userStats.length; +} + +/** + * Progress event emitted after each session file is fully processed. + * `current` is the number of files completed (skipped + parsed), + * `total` is the size of the work set. `processed` is the running total + * of inserted rows. + */ +export interface SyncProgress { + current: number; + total: number; + processed: number; + sessionFile: string; +} + +export interface SyncOptions { + /** Called after each file completes. Synchronous; keep it cheap. */ + onProgress?: (event: SyncProgress) => void; + /** + * Worker pool size. Defaults to a sensible value derived from the host + * (capped to avoid drowning a small machine in workers). Set to `1` to + * force serial parsing without spawning workers. + */ + workers?: number; +} + +function defaultWorkerCount(): number { + // `navigator.hardwareConcurrency` is the portable answer in Bun; fall + // back to a small fixed pool if it's somehow unavailable. + const hw = typeof navigator !== "undefined" ? (navigator.hardwareConcurrency ?? 0) : 0; + const raw = hw > 0 ? hw : 4; + // Cap at 8 - parse is JSON-bound, and SQLite writes serialize on main + // thread anyway, so more workers stop helping. + return Math.min(8, Math.max(2, Math.floor(raw))); +} + +interface WorkerHandle { + worker: Worker; + busy: boolean; + resolve: ((res: ParseSessionResult) => void) | null; + reject: ((err: Error) => void) | null; +} + +function spawnWorker(): WorkerHandle { + const worker = new Worker(syncWorkerUrl, { type: "module" }); + const handle: WorkerHandle = { worker, busy: false, resolve: null, reject: null }; + worker.onmessage = (event: MessageEvent) => { + const { resolve, reject } = handle; + handle.resolve = null; + handle.reject = null; + handle.busy = false; + if (!resolve || !reject) return; + if (event.data.ok) resolve(event.data.result); + else reject(new Error(event.data.error)); + }; + worker.onerror = (event: ErrorEvent) => { + const { reject } = handle; + handle.resolve = null; + handle.reject = null; + handle.busy = false; + reject?.(event.error instanceof Error ? event.error : new Error(event.message || "worker error")); + }; + return handle; +} + +function dispatch(handle: WorkerHandle, request: SyncWorkerRequest): Promise { + if (handle.busy) { + return Promise.reject(new Error("worker is busy - this is a bug in the dispatcher")); } - - const lastModified = fileStats.mtimeMs; - - // Check if file has changed since last sync - const stored = getFileOffset(sessionFile); - if (stored && stored.lastModified >= lastModified) { - return 0; // File hasn't changed - } - - // Parse file from last offset - const fromOffset = stored?.offset ?? 0; - const { stats, userStats, newOffset } = await parseSessionFile(sessionFile, fromOffset); - - if (stats.length > 0) { - insertMessageStats(stats); - } - if (userStats.length > 0) { - insertUserMessageStats(userStats); - } - - // Update offset tracker - setFileOffset(sessionFile, newOffset, lastModified); - - return stats.length + userStats.length; + const { promise, resolve, reject } = Promise.withResolvers(); + handle.busy = true; + handle.resolve = resolve; + handle.reject = reject; + handle.worker.postMessage(request); + return promise; } /** * Sync all session files to the database. - * Returns the number of new entries processed. + * + * Parsing fans out across a worker pool (one in-flight job per worker) + * while DB writes and offset bookkeeping stay on the calling thread so the + * single SQLite handle stays uncontended. `onProgress` fires once per + * completed file (skipped files included so the bar walks at a steady + * rate). */ -export async function syncAllSessions(): Promise<{ processed: number; files: number }> { +export async function syncAllSessions(opts?: SyncOptions): Promise<{ processed: number; files: number }> { await initDb(); const files = await listAllSessionFiles(); + if (files.length === 0) return { processed: 0, files: 0 }; + let totalProcessed = 0; let filesProcessed = 0; + let completed = 0; + let cursor = 0; - for (const file of files) { - const count = await syncSessionFile(file); - if (count > 0) { - totalProcessed += count; - filesProcessed++; + const poolSize = Math.max(1, Math.min(files.length, opts?.workers ?? defaultWorkerCount())); + const handles: WorkerHandle[] = []; + for (let i = 0; i < poolSize; i++) handles.push(spawnWorker()); + + const report = (sessionFile: string) => { + completed++; + opts?.onProgress?.({ + current: completed, + total: files.length, + processed: totalProcessed, + sessionFile, + }); + }; + + async function drain(handle: WorkerHandle): Promise { + while (true) { + const idx = cursor++; + if (idx >= files.length) return; + const sessionFile = files[idx]; + + let fileStats: fs.Stats; + try { + fileStats = await fs.promises.stat(sessionFile); + } catch { + report(sessionFile); + continue; + } + const lastModified = fileStats.mtimeMs; + const stored = getFileOffset(sessionFile); + if (stored && stored.lastModified >= lastModified) { + report(sessionFile); + continue; + } + + const fromOffset = stored?.offset ?? 0; + const result = await dispatch(handle, { sessionFile, fromOffset }); + const inserted = applyParseResult(sessionFile, lastModified, result); + if (inserted > 0) { + totalProcessed += inserted; + filesProcessed++; + } + report(sessionFile); } } + try { + await Promise.all(handles.map(drain)); + } finally { + for (const handle of handles) handle.worker.terminate(); + } + return { processed: totalProcessed, files: filesProcessed }; } diff --git a/packages/stats/src/client/components/BehaviorChart.tsx b/packages/stats/src/client/components/BehaviorChart.tsx index 7de80a42d..7bcf00e8e 100644 --- a/packages/stats/src/client/components/BehaviorChart.tsx +++ b/packages/stats/src/client/components/BehaviorChart.tsx @@ -51,10 +51,14 @@ const CHART_THEMES = { } as const; const METRIC_OPTIONS = [ - { value: "yellingSentences", label: "Yelling" }, + { value: "yelling", label: "Yelling" }, { value: "profanity", label: "Profanity" }, - { value: "dramaRuns", label: "Drama (!!! / ???)" }, - { value: "total", label: "All three combined" }, + { value: "anguish", label: "Anguish (!!!, nooo, dude, ..)" }, + { value: "negation", label: "Negation (no/nope/wrong)" }, + { value: "repetition", label: "Repetition (i meant, still doesnt)" }, + { value: "blame", label: "Blame (you didnt, stop X-ing)" }, + { value: "frustration", label: "Frustration (neg + rep + blame)" }, + { value: "total", label: "All signals combined" }, ] as const; type Metric = (typeof METRIC_OPTIONS)[number]["value"]; @@ -70,7 +74,10 @@ interface BehaviorChartProps { } function pointHits(point: BehaviorTimeSeriesPoint, metric: Metric): number { - if (metric === "total") return point.yellingSentences + point.profanity + point.dramaRuns; + if (metric === "frustration") return point.negation + point.repetition + point.blame; + if (metric === "total") { + return point.yelling + point.profanity + point.anguish + point.negation + point.repetition + point.blame; + } return point[metric]; } diff --git a/packages/stats/src/client/components/BehaviorModelsTable.tsx b/packages/stats/src/client/components/BehaviorModelsTable.tsx index 924b61173..d001f7949 100644 --- a/packages/stats/src/client/components/BehaviorModelsTable.tsx +++ b/packages/stats/src/client/components/BehaviorModelsTable.tsx @@ -30,7 +30,8 @@ const MODEL_COLORS = [ const SERIES_COLORS = { yelling: "#fbbf24", // amber profanity: "#f87171", // red - drama: "#a78bfa", // violet + anguish: "#a78bfa", // violet + frustration: "#22d3ee", // cyan - new semantic signals } as const; const CHART_THEMES = { @@ -65,7 +66,8 @@ interface DailyPoint { timestamp: number; yelling: number; profanity: number; - drama: number; + anguish: number; + frustration: number; total: number; } @@ -73,7 +75,7 @@ interface ModelTrendSeries { data: DailyPoint[]; } -const GRID_TEMPLATE = "2fr 0.9fr 0.9fr 0.9fr 0.9fr 0.9fr 140px 40px"; +const GRID_TEMPLATE = "2fr 0.9fr 0.8fr 0.8fr 0.8fr 0.9fr 0.8fr 140px 40px"; function formatInt(value: number): string { return value.toLocaleString(); @@ -81,7 +83,13 @@ function formatInt(value: number): string { function totalHitRate(model: BehaviorModelStats): number { if (model.totalMessages === 0) return 0; - const hits = model.totalYellingSentences + model.totalProfanity + model.totalDramaRuns; + const hits = + model.totalYelling + + model.totalProfanity + + model.totalAnguish + + model.totalNegation + + model.totalRepetition + + model.totalBlame; return hits / model.totalMessages; } @@ -128,7 +136,8 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
Messages
CAPS %
Profanity %
-
Drama %
+
Anguish %
+
Frustration %
Hits %
Trend
@@ -140,7 +149,8 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa const trend = trendByKey.get(key)?.data ?? []; const trendColor = MODEL_COLORS[index % MODEL_COLORS.length]; const isExpanded = expandedKey === key; - const totalHits = model.totalYellingSentences + model.totalProfanity + model.totalDramaRuns; + const totalFrustration = model.totalNegation + model.totalRepetition + model.totalBlame; + const totalHits = model.totalYelling + model.totalProfanity + model.totalAnguish + totalFrustration; return (
@@ -158,13 +168,16 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa {formatInt(model.totalMessages)}
- {formatRate(model.totalYellingSentences, model.totalMessages)} + {formatRate(model.totalYelling, model.totalMessages)}
{formatRate(model.totalProfanity, model.totalMessages)}
- {formatRate(model.totalDramaRuns, model.totalMessages)} + {formatRate(model.totalAnguish, model.totalMessages)} +
+
+ {formatRate(totalFrustration, model.totalMessages)}
{formatRate(totalHits, model.totalMessages)} @@ -188,7 +201,7 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
@@ -199,11 +212,29 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa valueClass="text-[var(--accent-red,#f87171)]" /> + + + d.drama), - borderColor: SERIES_COLORS.drama, + label: "Anguish", + data: data.map(d => d.anguish), + borderColor: SERIES_COLORS.anguish, + backgroundColor: "transparent", + tension: 0.4, + pointRadius: 0, + borderWidth: 2, + }, + { + label: "Frustration", + data: data.map(d => d.frustration), + borderColor: SERIES_COLORS.frustration, backgroundColor: "transparent", tension: 0.4, pointRadius: 0, @@ -394,13 +434,15 @@ function buildTrendLookup(points: BehaviorTimeSeriesPoint[]): Map { const totals = new Map(); for (const point of behaviorSeries) { const key = `${point.model}::${point.provider}`; const existing = totals.get(key); - const score = point.yellingSentences + point.profanity + point.dramaRuns; + const score = + point.yelling + point.profanity + point.anguish + point.negation + point.repetition + point.blame; if (existing) { existing.score += score; } else { @@ -31,36 +43,44 @@ export function BehaviorSummary({ overall, behaviorSeries }: BehaviorSummaryProp return best; }, [behaviorSeries]); - const capsPerMsg = overall.totalMessages > 0 ? overall.totalYellingSentences / overall.totalMessages : 0; + const totalFrustration = overall.totalNegation + overall.totalRepetition + overall.totalBlame; + const messages = overall.totalMessages; const cards: Array<{ label: string; value: string; sub?: string }> = [ { label: "Messages", value: formatInt(overall.totalMessages), + sub: messages > 0 ? "in selected range" : undefined, }, { label: "Yelling", - value: formatInt(overall.totalYellingSentences), - sub: overall.totalMessages > 0 ? `${capsPerMsg.toFixed(2)} / msg` : undefined, + value: formatInt(overall.totalYelling), + sub: perMsg(overall.totalYelling, messages), }, { label: "Profanity hits", value: formatInt(overall.totalProfanity), + sub: perMsg(overall.totalProfanity, messages), }, { - label: "Drama runs", - value: formatInt(overall.totalDramaRuns), - sub: "!!! / ???", + label: "Anguish", + value: formatInt(overall.totalAnguish), + sub: perMsg(overall.totalAnguish, messages), + }, + { + label: "Frustration", + value: formatInt(totalFrustration), + sub: perMsg(totalFrustration, messages), }, { label: "Most yelled-at", - value: topModel?.model ?? "—", + value: topModel?.model ?? "\u2014", sub: topModel ? `${formatInt(topModel.score)} hits` : undefined, }, ]; return ( -
+
{cards.map(card => (

{card.label}

diff --git a/packages/stats/src/client/types.ts b/packages/stats/src/client/types.ts index 0b338c0ae..ab783d5c7 100644 --- a/packages/stats/src/client/types.ts +++ b/packages/stats/src/client/types.ts @@ -135,17 +135,23 @@ export interface BehaviorTimeSeriesPoint { model: string; provider: string; messages: number; - yellingSentences: number; + yelling: number; profanity: number; - dramaRuns: number; + anguish: number; + negation: number; + repetition: number; + blame: number; chars: number; } export interface BehaviorOverallStats { totalMessages: number; - totalYellingSentences: number; + totalYelling: number; totalProfanity: number; - totalDramaRuns: number; + totalAnguish: number; + totalNegation: number; + totalRepetition: number; + totalBlame: number; totalChars: number; firstTimestamp: number; lastTimestamp: number; @@ -155,9 +161,12 @@ export interface BehaviorModelStats { model: string; provider: string; totalMessages: number; - totalYellingSentences: number; + totalYelling: number; totalProfanity: number; - totalDramaRuns: number; + totalAnguish: number; + totalNegation: number; + totalRepetition: number; + totalBlame: number; totalChars: number; lastTimestamp: number; } diff --git a/packages/stats/src/db.ts b/packages/stats/src/db.ts index 41316e3de..659f9b988 100644 --- a/packages/stats/src/db.ts +++ b/packages/stats/src/db.ts @@ -14,6 +14,7 @@ import type { ModelStats, ModelTimeSeriesPoint, TimeSeriesPoint, + UserMessageLink, UserMessageStats, } from "./types"; @@ -98,9 +99,12 @@ export async function initDb(): Promise { provider TEXT, chars INTEGER NOT NULL, words INTEGER NOT NULL, - yelling_sentences INTEGER NOT NULL, + yelling INTEGER NOT NULL, profanity INTEGER NOT NULL, - drama_runs INTEGER NOT NULL, + anguish INTEGER NOT NULL, + negation INTEGER NOT NULL DEFAULT 0, + repetition INTEGER NOT NULL DEFAULT 0, + blame INTEGER NOT NULL DEFAULT 0, UNIQUE(session_file, entry_id) ); @@ -118,16 +122,30 @@ export async function initDb(): Promise { db.exec("ALTER TABLE messages ADD COLUMN premium_requests REAL NOT NULL DEFAULT 0"); } db.exec("UPDATE messages SET premium_requests = 0 WHERE premium_requests IS NULL"); - // Bumping the metric definition (yelling sentences vs caps words) invalidates - // previously-ingested rows. If the legacy column is present we drop the table - // outright; the `IF NOT EXISTS` create above already gave us the new schema - // in parallel, but we want a clean wipe + re-ingest. The accompanying - // `backfillUserMessages` bump (v2) clears `file_offsets` so the next sync - // re-parses every session. + // Each behavior-metric bump invalidates previously-ingested rows. We detect + // the stale schema by column name and drop the table; `IF NOT EXISTS` above + // already produced the new schema, but we want a clean wipe + re-ingest. + // `backfillUserMessages` then clears `file_offsets` so the next sync + // re-parses every session under the current metric definitions. + // v1 -> v2: yelling sentences replace `caps_words`. + // v2 -> v3: `drama_runs` folded into a single `anguish` signal that + // also captures elongated interjections, `dude`, and dot runs, + // gated on a stripped prose-line budget. + // v3 -> v4: added `negation`, `repetition`, `blame` frustration signals + // plus profanity dictionary expansion + word-boundary fix. + // v4 -> v5: column `yelling_sentences` renamed to `yelling` to match + // the other single-word signal columns. const userMessageColumns = db.prepare("PRAGMA table_info(user_messages)").all() as { name: string; }[]; - if (userMessageColumns.some(column => column.name === "caps_words")) { + const hasStaleColumn = + userMessageColumns.length > 0 && + (userMessageColumns.some(column => column.name === "caps_words") || + userMessageColumns.some(column => column.name === "drama_runs") || + userMessageColumns.some(column => column.name === "yelling_sentences")); + const hasV4Columns = userMessageColumns.some(column => column.name === "negation"); + const hasOldUserMessages = userMessageColumns.length > 0; + if (hasStaleColumn || (hasOldUserMessages && !hasV4Columns)) { db.exec("DROP TABLE user_messages"); db.exec(` CREATE TABLE user_messages ( @@ -140,9 +158,12 @@ export async function initDb(): Promise { provider TEXT, chars INTEGER NOT NULL, words INTEGER NOT NULL, - yelling_sentences INTEGER NOT NULL, + yelling INTEGER NOT NULL, profanity INTEGER NOT NULL, - drama_runs INTEGER NOT NULL, + anguish INTEGER NOT NULL, + negation INTEGER NOT NULL DEFAULT 0, + repetition INTEGER NOT NULL DEFAULT 0, + blame INTEGER NOT NULL DEFAULT 0, UNIQUE(session_file, entry_id) ); CREATE INDEX IF NOT EXISTS idx_user_messages_timestamp ON user_messages(timestamp); @@ -150,6 +171,7 @@ export async function initDb(): Promise { `); } backfillUserMessages(db); + repairUserMessageLinks(db); backfillMissingCatalogCosts(db); return db; } @@ -704,11 +726,19 @@ export function getCostTimeSeries(days = 90, cutoff?: number | null): CostTimeSe * - v1: initial introduction of `user_messages`. * - v2: yelling-sentence metric replaces caps-word counts; existing rows are * computed under the old definition and must be discarded. + * - v3: drama runs collapsed into `anguish` (drama + elongated interjections + * + `dude` + dot runs), scored on a stripped prose body and gated on + * line count. Existing rows used the narrower definition. + * - v4: added `negation` / `repetition` / `blame` signals and fixed a + * latent word-boundary bug in the profanity / anguish regexes that had + * left those metrics matching nothing in real prose. + * - v5: renamed `yelling_sentences` column to `yelling` to match the other + * single-word signal columns (profanity, anguish, negation, ...). * - * Existing `messages` rows are unaffected — `INSERT OR IGNORE` keeps them. + * Existing `messages` rows are unaffected - `INSERT OR IGNORE` keeps them. */ function backfillUserMessages(database: Database): void { - const row = database.prepare("SELECT value FROM meta WHERE key = 'user_messages_v2'").get() as + const row = database.prepare("SELECT value FROM meta WHERE key = 'user_messages_v5'").get() as | { value: string } | undefined; if (row) return; @@ -717,7 +747,27 @@ function backfillUserMessages(database: Database): void { database.exec("DELETE FROM file_offsets"); database .prepare("INSERT OR REPLACE INTO meta (key, value) VALUES (?, ?)") - .run("user_messages_v2", String(Date.now())); + .run("user_messages_v5", String(Date.now())); +} + +/** + * One-shot wipe of `file_offsets` to force `parseSessionFile` to re-parse + * every session from byte zero. We don't touch `user_messages`; the parser + * now emits a `UserMessageLink` for every assistant->parent pair, and the + * guarded `updateUserMessageLinks` UPDATE fixes any row whose `model` was + * left NULL by the old in-pass-only linking logic. Idempotent: gated by a + * sentinel row in `meta`. + */ +function repairUserMessageLinks(database: Database): void { + const row = database.prepare("SELECT value FROM meta WHERE key = 'user_message_links_v1'").get() as + | { value: string } + | undefined; + if (row) return; + + database.exec("DELETE FROM file_offsets"); + database + .prepare("INSERT OR REPLACE INTO meta (key, value) VALUES (?, ?)") + .run("user_message_links_v1", String(Date.now())); } /** @@ -729,8 +779,9 @@ export function insertUserMessageStats(stats: UserMessageStats[]): number { const stmt = db.prepare(` INSERT OR IGNORE INTO user_messages ( session_file, entry_id, folder, timestamp, model, provider, - chars, words, yelling_sentences, profanity, drama_runs - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + chars, words, yelling, profanity, anguish, + negation, repetition, blame + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) `); let inserted = 0; @@ -745,9 +796,12 @@ export function insertUserMessageStats(stats: UserMessageStats[]): number { s.provider, s.chars, s.words, - s.yellingSentences, + s.yelling, s.profanity, - s.dramaRuns, + s.anguish, + s.negation, + s.repetition, + s.blame, ); if (result.changes > 0) inserted++; } @@ -756,6 +810,35 @@ export function insertUserMessageStats(stats: UserMessageStats[]): number { return inserted; } +/** + * Backfill the responding `model`/`provider` on user-message rows that were + * persisted before their assistant reply was parsed (a side effect of + * incremental `fromOffset` syncing: the `userByEntryId` map in + * `parseSessionFile` only spans a single pass). Each row is updated at most + * once because the `model IS NULL` guard short-circuits subsequent passes. + * + * Returns the number of rows actually updated. + */ +export function updateUserMessageLinks(links: UserMessageLink[]): number { + if (!db || links.length === 0) return 0; + + const stmt = db.prepare(` + UPDATE user_messages + SET model = ?, provider = ? + WHERE session_file = ? AND entry_id = ? AND model IS NULL + `); + + let updated = 0; + const apply = db.transaction(() => { + for (const link of links) { + const result = stmt.run(link.model, link.provider, link.sessionFile, link.entryId); + if (result.changes > 0) updated++; + } + }); + apply(); + return updated; +} + const UNKNOWN_MODEL = "unknown"; interface BehaviorSeriesRow { @@ -763,9 +846,12 @@ interface BehaviorSeriesRow { model: string; provider: string; messages: number; - yelling_sentences: number | null; + yelling: number | null; profanity: number | null; - drama_runs: number | null; + anguish: number | null; + negation: number | null; + repetition: number | null; + blame: number | null; chars: number | null; } @@ -781,9 +867,12 @@ export function getBehaviorTimeSeries(cutoff?: number | null): BehaviorTimeSerie COALESCE(model, ?) as model, COALESCE(provider, ?) as provider, COUNT(*) as messages, - SUM(yelling_sentences) as yelling_sentences, + SUM(yelling) as yelling, SUM(profanity) as profanity, - SUM(drama_runs) as drama_runs, + SUM(anguish) as anguish, + SUM(negation) as negation, + SUM(repetition) as repetition, + SUM(blame) as blame, SUM(chars) as chars FROM user_messages ${hasCutoff ? "WHERE timestamp >= ?" : ""} @@ -798,18 +887,24 @@ export function getBehaviorTimeSeries(cutoff?: number | null): BehaviorTimeSerie model: row.model, provider: row.provider, messages: row.messages, - yellingSentences: row.yelling_sentences ?? 0, + yelling: row.yelling ?? 0, profanity: row.profanity ?? 0, - dramaRuns: row.drama_runs ?? 0, + anguish: row.anguish ?? 0, + negation: row.negation ?? 0, + repetition: row.repetition ?? 0, + blame: row.blame ?? 0, chars: row.chars ?? 0, })); } interface BehaviorOverallRow { total_messages: number; - total_yelling_sentences: number | null; + total_yelling: number | null; total_profanity: number | null; - total_drama_runs: number | null; + total_anguish: number | null; + total_negation: number | null; + total_repetition: number | null; + total_blame: number | null; total_chars: number | null; first_timestamp: number | null; last_timestamp: number | null; @@ -821,9 +916,12 @@ interface BehaviorOverallRow { export function getBehaviorOverall(cutoff?: number | null): BehaviorOverallStats { const empty: BehaviorOverallStats = { totalMessages: 0, - totalYellingSentences: 0, + totalYelling: 0, totalProfanity: 0, - totalDramaRuns: 0, + totalAnguish: 0, + totalNegation: 0, + totalRepetition: 0, + totalBlame: 0, totalChars: 0, firstTimestamp: 0, lastTimestamp: 0, @@ -833,9 +931,12 @@ export function getBehaviorOverall(cutoff?: number | null): BehaviorOverallStats const stmt = db.prepare(` SELECT COUNT(*) as total_messages, - SUM(yelling_sentences) as total_yelling_sentences, + SUM(yelling) as total_yelling, SUM(profanity) as total_profanity, - SUM(drama_runs) as total_drama_runs, + SUM(anguish) as total_anguish, + SUM(negation) as total_negation, + SUM(repetition) as total_repetition, + SUM(blame) as total_blame, SUM(chars) as total_chars, MIN(timestamp) as first_timestamp, MAX(timestamp) as last_timestamp @@ -846,9 +947,12 @@ export function getBehaviorOverall(cutoff?: number | null): BehaviorOverallStats if (!row?.total_messages) return empty; return { totalMessages: row.total_messages, - totalYellingSentences: row.total_yelling_sentences ?? 0, + totalYelling: row.total_yelling ?? 0, totalProfanity: row.total_profanity ?? 0, - totalDramaRuns: row.total_drama_runs ?? 0, + totalAnguish: row.total_anguish ?? 0, + totalNegation: row.total_negation ?? 0, + totalRepetition: row.total_repetition ?? 0, + totalBlame: row.total_blame ?? 0, totalChars: row.total_chars ?? 0, firstTimestamp: row.first_timestamp ?? 0, lastTimestamp: row.last_timestamp ?? 0, @@ -859,9 +963,12 @@ interface BehaviorByModelRow { model: string; provider: string; total_messages: number; - total_yelling_sentences: number | null; + total_yelling: number | null; total_profanity: number | null; - total_drama_runs: number | null; + total_anguish: number | null; + total_negation: number | null; + total_repetition: number | null; + total_blame: number | null; total_chars: number | null; last_timestamp: number | null; } @@ -878,9 +985,12 @@ export function getBehaviorByModel(cutoff?: number | null): BehaviorModelStats[] COALESCE(model, ?) as model, COALESCE(provider, ?) as provider, COUNT(*) as total_messages, - SUM(yelling_sentences) as total_yelling_sentences, + SUM(yelling) as total_yelling, SUM(profanity) as total_profanity, - SUM(drama_runs) as total_drama_runs, + SUM(anguish) as total_anguish, + SUM(negation) as total_negation, + SUM(repetition) as total_repetition, + SUM(blame) as total_blame, SUM(chars) as total_chars, MAX(timestamp) as last_timestamp FROM user_messages @@ -895,9 +1005,12 @@ export function getBehaviorByModel(cutoff?: number | null): BehaviorModelStats[] model: row.model, provider: row.provider, totalMessages: row.total_messages, - totalYellingSentences: row.total_yelling_sentences ?? 0, + totalYelling: row.total_yelling ?? 0, totalProfanity: row.total_profanity ?? 0, - totalDramaRuns: row.total_drama_runs ?? 0, + totalAnguish: row.total_anguish ?? 0, + totalNegation: row.total_negation ?? 0, + totalRepetition: row.total_repetition ?? 0, + totalBlame: row.total_blame ?? 0, totalChars: row.total_chars ?? 0, lastTimestamp: row.last_timestamp ?? 0, })); diff --git a/packages/stats/src/index.ts b/packages/stats/src/index.ts index d72f05127..862308ee6 100755 --- a/packages/stats/src/index.ts +++ b/packages/stats/src/index.ts @@ -6,7 +6,13 @@ import { getDashboardStats, getTotalMessageCount, syncAllSessions } from "./aggr import { closeDb } from "./db"; import { startServer } from "./server"; -export { getDashboardStats, getTotalMessageCount, syncAllSessions } from "./aggregator"; +export { + getDashboardStats, + getTotalMessageCount, + type SyncOptions, + type SyncProgress, + syncAllSessions, +} from "./aggregator"; export { closeDb } from "./db"; export { startServer } from "./server"; export type { @@ -112,8 +118,28 @@ Examples: try { // Sync first - console.log("Syncing session files..."); - const { processed, files } = await syncAllSessions(); + const tty = process.stderr.isTTY === true; + process.stderr.write("Syncing session files...\n"); + let lastWidth = 0; + let lastRender = 0; + const { processed, files } = await syncAllSessions({ + onProgress: event => { + if (!tty) return; + const now = Date.now(); + if (event.current < event.total && now - lastRender < 33) return; + lastRender = now; + const marker = "/sessions/"; + const idx = event.sessionFile.indexOf(marker); + const short = idx >= 0 ? event.sessionFile.slice(idx + marker.length) : event.sessionFile; + const pct = ((event.current / event.total) * 100).toFixed(0).padStart(3, " "); + const line = `[${event.current}/${event.total}] ${pct}% ${short}`; + const columns = process.stderr.columns ?? 120; + const clipped = line.length > columns - 1 ? `${line.slice(0, columns - 2)}\u2026` : line; + process.stderr.write(`\r${clipped.padEnd(lastWidth)}`); + lastWidth = clipped.length; + }, + }); + if (tty && lastWidth > 0) process.stderr.write(`\r${" ".repeat(lastWidth)}\r`); const total = await getTotalMessageCount(); console.log(`Synced ${processed} new entries from ${files} files (${total} total)\n`); diff --git a/packages/stats/src/parser.ts b/packages/stats/src/parser.ts index 3306b0186..ff7f2fc18 100644 --- a/packages/stats/src/parser.ts +++ b/packages/stats/src/parser.ts @@ -2,7 +2,7 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; import type { AssistantMessage } from "@oh-my-pi/pi-ai"; import { getSessionsDir, isEnoent } from "@oh-my-pi/pi-utils"; -import type { MessageStats, SessionEntry, SessionMessageEntry, UserMessageStats } from "./types"; +import type { MessageStats, SessionEntry, SessionMessageEntry, UserMessageLink, UserMessageStats } from "./types"; import { computeUserMessageMetrics } from "./user-metrics"; /** @@ -71,9 +71,12 @@ function extractUserStats(sessionFile: string, folder: string, entry: SessionMes provider: null, chars: metrics.chars, words: metrics.words, - yellingSentences: metrics.yellingSentences, + yelling: metrics.yelling, profanity: metrics.profanity, - dramaRuns: metrics.dramaRuns, + anguish: metrics.anguish, + negation: metrics.negation, + repetition: metrics.repetition, + blame: metrics.blame, }; } @@ -131,21 +134,26 @@ function parseSessionEntriesLenient(bytes: Uint8Array): { entries: SessionEntry[ * Parse a session file and extract all assistant message stats. * Uses incremental reading with offset tracking. */ -export async function parseSessionFile( - sessionPath: string, - fromOffset = 0, -): Promise<{ stats: MessageStats[]; userStats: UserMessageStats[]; newOffset: number }> { +export interface ParseSessionResult { + stats: MessageStats[]; + userStats: UserMessageStats[]; + userLinks: UserMessageLink[]; + newOffset: number; +} + +export async function parseSessionFile(sessionPath: string, fromOffset = 0): Promise { let bytes: Uint8Array; try { bytes = await Bun.file(sessionPath).bytes(); } catch (err) { - if (isEnoent(err)) return { stats: [], userStats: [], newOffset: fromOffset }; + if (isEnoent(err)) return { stats: [], userStats: [], userLinks: [], newOffset: fromOffset }; throw err; } const folder = extractFolderFromPath(sessionPath); const stats: MessageStats[] = []; const userStats: UserMessageStats[] = []; + const userLinks: UserMessageLink[] = []; const userByEntryId = new Map(); const start = Math.max(0, Math.min(fromOffset, bytes.length)); const unprocessed = bytes.subarray(start); @@ -165,17 +173,26 @@ export async function parseSessionFile( // Link assistant's responding model back to the user message it answered. const parentId = (entry as SessionMessageEntry).parentId; if (parentId) { - const parentUser = userByEntryId.get(parentId); - if (parentUser && parentUser.model === null) { - const msg = entry.message as AssistantMessage; - parentUser.model = msg.model; - parentUser.provider = msg.provider; + const msg = entry.message as AssistantMessage; + if (msg.model && msg.provider) { + // Emit unconditionally. The aggregator's UPDATE is guarded by + // `model IS NULL` so this is idempotent: a no-op for already + // linked rows, a fix-up for fresh inserts (which start NULL + // because the user row is recorded before its reply lands) and + // for cross-pass orphans whose parent was committed by an + // earlier incremental sync. + userLinks.push({ + sessionFile: sessionPath, + entryId: parentId, + model: msg.model, + provider: msg.provider, + }); } } } } - return { stats, userStats, newOffset: start + read }; + return { stats, userStats, userLinks, newOffset: start + read }; } /** diff --git a/packages/stats/src/sync-worker.ts b/packages/stats/src/sync-worker.ts new file mode 100644 index 000000000..52b3d7961 --- /dev/null +++ b/packages/stats/src/sync-worker.ts @@ -0,0 +1,31 @@ +/** + * Stateless parse worker for `syncAllSessions`. The main thread owns the + * SQLite handle; workers receive `{ sessionFile, fromOffset }`, run + * `parseSessionFile` (which is pure I/O + CPU, no DB), and post the + * structured-clone-safe result back. One in-flight request per worker so + * the main thread can fan jobs out 1:1 with the pool size. + */ + +import { type ParseSessionResult, parseSessionFile } from "./parser"; + +export interface SyncWorkerRequest { + sessionFile: string; + fromOffset: number; +} + +export type SyncWorkerResponse = { ok: true; result: ParseSessionResult } | { ok: false; error: string }; + +declare const self: Worker & { + onmessage: ((event: MessageEvent) => void) | null; +}; + +self.onmessage = async event => { + const { sessionFile, fromOffset } = event.data; + try { + const result = await parseSessionFile(sessionFile, fromOffset); + self.postMessage({ ok: true, result } satisfies SyncWorkerResponse); + } catch (err) { + const error = err instanceof Error ? (err.stack ?? err.message) : String(err); + self.postMessage({ ok: false, error } satisfies SyncWorkerResponse); + } +}; diff --git a/packages/stats/src/types.ts b/packages/stats/src/types.ts index 2b85492a5..a7b0dc88e 100644 --- a/packages/stats/src/types.ts +++ b/packages/stats/src/types.ts @@ -219,11 +219,31 @@ export interface UserMessageStats { /** Whitespace-delimited word count */ words: number; /** Yelling sentences (> 50% uppercase letters) */ - yellingSentences: number; + yelling: number; /** Profanity hits */ profanity: number; - /** Runs of 3+ consecutive `!` / `?` */ - dramaRuns: number; + /** Catch-all upset signal: drama runs + `noooo`/`ughh`/... + `dude` + `..` */ + anguish: number; + /** Corrective negation ("no", "nope", "thats not what i meant") */ + negation: number; + /** User repeating themselves ("i meant", "still doesnt work", "like i said") */ + repetition: number; + /** Second-person reproach ("you didnt", "you broke", "stop X-ing") */ + blame: number; +} + +/** + * Pair emitted by the parser when it sees an assistant message whose + * `parentId` points to a user message that wasn't parsed in the same pass + * (e.g. user prompt landed in an earlier incremental sync). The aggregator + * applies the link to the persisted `user_messages` row so it stops showing + * up in the "unknown" model bucket. + */ +export interface UserMessageLink { + sessionFile: string; + entryId: string; + model: string; + provider: string; } /** @@ -239,20 +259,29 @@ export interface BehaviorTimeSeriesPoint { /** Number of user messages in bucket */ messages: number; /** Total yelling sentences in bucket */ - yellingSentences: number; + yelling: number; /** Total profanity hits in bucket */ profanity: number; - /** Total drama runs in bucket */ - dramaRuns: number; + /** Total anguish signal in bucket */ + anguish: number; + /** Total corrective-negation hits in bucket */ + negation: number; + /** Total user-repeating-themselves hits in bucket */ + repetition: number; + /** Total second-person blame hits in bucket */ + blame: number; /** Total characters in bucket */ chars: number; } export interface BehaviorOverallStats { totalMessages: number; - totalYellingSentences: number; + totalYelling: number; totalProfanity: number; - totalDramaRuns: number; + totalAnguish: number; + totalNegation: number; + totalRepetition: number; + totalBlame: number; totalChars: number; firstTimestamp: number; lastTimestamp: number; @@ -265,9 +294,12 @@ export interface BehaviorModelStats { model: string; provider: string; totalMessages: number; - totalYellingSentences: number; + totalYelling: number; totalProfanity: number; - totalDramaRuns: number; + totalAnguish: number; + totalNegation: number; + totalRepetition: number; + totalBlame: number; totalChars: number; lastTimestamp: number; } diff --git a/packages/stats/src/user-metrics.ts b/packages/stats/src/user-metrics.ts index 750ed3e36..8eb737c5f 100644 --- a/packages/stats/src/user-metrics.ts +++ b/packages/stats/src/user-metrics.ts @@ -13,13 +13,55 @@ export interface UserMessageMetrics { /** * Number of "yelling" sentences: sentences where more than half of the * alphabetic characters are uppercase (and there are enough letters to - * make the ratio meaningful — short acronyms like "OK" don't count). + * make the ratio meaningful - short acronyms like "OK" don't count). */ - yellingSentences: number; + yelling: number; /** Profanity hits (word-boundary, case-insensitive). */ profanity: number; - /** Runs of 3+ `!` / `?` characters (including `1`-mishit fallout). */ - dramaRuns: number; + /** + * Catch-all "obviously upset" signal computed on a *prose-only* body + * (code fences, XML/HTML tags, URLs, file mentions, and quoted lines + * are stripped first; messages whose remaining prose is >=3 lines score + * zero because formatted prompts aren't tantrums). + * + * Sum of: + * - drama runs: 3+ `!` / `?` (with `1`-mishit fallout) + * - elongated interjections: `noooo`, `ahhhh`, `ughhh`, `argh`, `stooop`, + * `whyyy`, `fuuu(ck)`, `shiiit`, `wtfff`, `omggg`, `yessss`, `helpp`, + * `goddd`, `dammm`, `bruhh` + * - standalone `dude` + * - dot runs: `..`, `...`, `....+` + */ + anguish: number; + /** + * Corrective negation: the user is telling us we got it wrong. + * + * Counted on the same prose-only body as {@link anguish}. + * + * - line-leading `no` / `nope` / `nah` / `nvm` / `wrong` / `incorrect` + * (word-bounded, so `now`, `nobody`, `north` don't match) + * - `that(?:'s)? not (what|right|it)` and `not what i (meant|asked|said|wanted)` + */ + negation: number; + /** + * The user is repeating themselves - strong signal the previous turn + * missed the ask. Counts hits for: + * + * - `i (meant|said|asked|told you|already (said|told|did|asked|wrote))` + * - `(like|as) i (said|told you|asked)` + * - `still (doesn't|isn't|not|broken|wrong|fails|failing|the same|same)` + * + * Bare `still` / `again` are too ambiguous to count alone (they show up + * in normal speech like "try again" or "still works"). + */ + repetition: number; + /** + * Direct second-person reproach pinned on the agent: + * + * - `you (didn't|did not|broke|missed|forgot|keep|always|never|still|ignored)` + * - sentence-leading `stop ing` imperatives + */ + blame: number; } /** @@ -363,15 +405,20 @@ const PROFANITY: readonly string[] = [ "garbage", "crud", "crudded", + // quality-dismissal ("this is garbage / pointless") + "useless", + "pointless", + "horrible", + "awful", + "worthless", + "ridiculous", + "nonsense", // religious exclamations "jesus", "christ", "jeez", "jeezus", "sheesh", - "holymoly", - "holyfuck", - "holysmokes", "godsake", // chat acronyms "wtf", @@ -415,18 +462,98 @@ const PROFANITY: readonly string[] = [ "grrrr", ]; -const PROFANITY_RE = new RegExp(`\\b(?:${PROFANITY.join("|")})\\b`, "gi"); +const PROFANITY_RE = new RegExp(String.raw`\b(?:${PROFANITY.join("|")})\b`, "gi"); const SENTENCE_RE = /[^.!?\n]+/g; const LETTER_RE = /\p{L}/gu; const UPPER_LETTER_RE = /\p{Lu}/gu; const YELLING_MIN_LETTERS = 4; const YELLING_THRESHOLD = 0.5; -// Runs starting with `!` or `?` followed by ≥2 of `!?1`. The `1` is the +// Runs starting with `!` or `?` followed by 2+ of `!?1`. The `1` is the // classic shift-key mishit ("!!!111" / "!?!??111") so we count those as // part of the same drama burst. const DRAMA_RE = /[!?][!?1]{2,}/g; const WORD_RE = /\S+/g; +// Elongated anguish/exasperation interjections. Each alternative is a +// case-insensitive word-bounded pattern that requires *real* elongation +// (so plain "no" / "argh" / "ahh" / "god" don't fire). Picked to avoid +// hex / base64 contamination via the surrounding `\b` plus letter-only +// alternatives. +const ANGUISH_PATTERNS: readonly string[] = [ + "no{3,}", // nooo, noooooo + "a+h{2,}", // ahh, aaaahhh + "u+g+h{2,}", // ughh, uuugh + "a+r+g+h+", // argh, aaargh, arrgghhh + "st+o{3,}p+", // stooop, sttooopp + "w+h+y{3,}", // whyyy, whyyyyy + "f+u{3,}c*k*", // fuuu, fuuuck + "wtf{3,}", // wtfff + "o+m+g{2,}", // omgg, omggg + "ye+s{3,}", // yesss, yeessss + "g+o+d{3,}", // goddd, goddddd + "br+u+h{2,}", // bruhh, bruuuhh +]; +const ANGUISH_RE = new RegExp(String.raw`\b(?:${ANGUISH_PATTERNS.join("|")})\b`, "gi"); +const DUDE_RE = /\bdude\b/gi; +// Runs of 2+ dots. Captures `..` (lazy trail-off), `...` (tentative +// ellipsis), and `....+` (exasperation) in a single signal. +const ELLIPSIS_RE = /\.{2,}/g; + +// --- Frustration signals ---------------------------------------------------- +// Each set of patterns below is tuned against ~42k real user prompts so the +// short-prose hits are dominated by genuine frustration, not technical talk. + +// Corrective negation. We deliberately anchor to the very start of the +// trimmed prose body (no `m` flag) - in practice mid-message lines that +// start with `no`/`Wrong`/`No JSDoc warning` are list items, pasted error +// text or descriptive statements, not actual corrections. Real frustration +// negation overwhelmingly opens the message. +const NEGATION_LEAD_RE = /^[ \t]*(?:no|nope|nah|nvm|wrong|incorrect)\b/gi; +const NEGATION_PHRASE_RE = + /\b(?:that['\u2019]?s\s+not\s+(?:what|right|it)|not\s+what\s+i\s+(?:meant|asked|said|wanted))\b/gi; + +// User repeating themselves. The recall pattern accepts an optional +// `like ` / `as ` prefix so "like i said" doesn't double-count with bare +// "i said". Bare `i asked` is too noisy - it's overwhelmingly "i asked +// " in this corpus (committee, experts, weaker LLM, ...) - +// so we require `i asked you` for that variant. Bare `still` / `again` are +// ambiguous so we only count `still` when followed by a negative or +// sameness marker. +const REPETITION_RECALL_RE = + /\b(?:(?:like|as)\s+i\s+(?:said|told\s+you|asked)|i\s+(?:meant|said|told\s+you|asked\s+you|already\s+(?:said|told|did|asked|wrote)))\b/gi; +const REPETITION_STILL_RE = + /\bstill\s+(?:doesn['\u2019]?t|doesnt|isn['\u2019]?t|isnt|not|broken|wrong|fails|failing|the\s+same|same)\b/gi; + +// Direct second-person reproach. `you` alone is too generic (>7k hits in +// short prose), so we anchor it to a small set of accusatory verbs. +const BLAME_YOU_RE = /\byou\s+(?:didn['\u2019]?t|did\s+not|broke|missed|forgot|keep|always|never|still|ignored)\b/gi; +// `stop ing` is only frustration when it's an imperative - require it +// to start a sentence (line start or after a sentence-terminating punctuator). +const BLAME_STOP_RE = /(?:^|(?<=[.!?\n]))\s*stop\s+\w+ing\b/gim; + +// Stripped from the analyzed body before scoring so that structured +// content (code, XML/HTML, URLs, file mentions, quoted blocks) doesn't +// pollute behavior signals. We replace with a newline so line counts +// reflect what was removed instead of merging neighbors. +const FENCED_CODE_RE = /```[\s\S]*?```/g; +const XML_TAG_PAIR_RE = /<([A-Za-z][\w-]*)\b[^>]*>[\s\S]*?<\/\1>/g; +const XML_TAG_BARE_RE = /<\/?[A-Za-z][\w-]*\b[^>]*\/?>/g; +const INLINE_CODE_RE = /`[^`\n]*`/g; +const URL_RE = /\bhttps?:\/\/\S+/gi; +const FILE_MENTION_RE = /(^|\s)@[\w./-]+/g; +const QUOTE_LINE_RE = /^[ \t]*>.*$/gm; +// Harness placeholders the TUI substitutes for binary/non-text user input. +// Strip them so real frustration signals on later lines aren't masked off +// by `[Image #1]` etc. consuming line 1. +const IMAGE_MARKER_RE = /\[Image #\d+\]/g; +// ANSI escape sequences sometimes leak in from terminal copy-paste +// (e.g. when the user pastes a bash transcript). Strip them. +const ANSI_ESCAPE_RE = /\x1b\[[0-9;]*[A-Za-z]/g; + +// Users don't really get angry with super detailed and formatted prompts +// - if the remaining prose is this many lines or more, score zero. +const MAX_PROSE_LINES = 3; + /** Count regex hits without materializing the match array. */ function countMatches(text: string, re: RegExp): number { let count = 0; @@ -457,6 +584,33 @@ function countYellingSentences(text: string): number { return count; } +/** + * Strip structured content so that pasted code, harness wrappers, file + * mentions and quoted blocks don't dilute or fake behavior signals. + * Each strip is replaced with a newline so subsequent line counting + * reflects what was removed instead of merging neighbors. + */ +function stripStructuredContent(text: string): string { + return text + .replace(FENCED_CODE_RE, "\n") + .replace(XML_TAG_PAIR_RE, "\n") + .replace(XML_TAG_BARE_RE, " ") + .replace(INLINE_CODE_RE, " ") + .replace(URL_RE, " ") + .replace(FILE_MENTION_RE, "$1 ") + .replace(QUOTE_LINE_RE, "") + .replace(IMAGE_MARKER_RE, " ") + .replace(ANSI_ESCAPE_RE, ""); +} + +function countNonEmptyLines(text: string): number { + let count = 0; + for (const line of text.split("\n")) { + if (line.trim().length > 0) count++; + } + return count; +} + /** * Compute behavioral metrics for a user message. * @@ -465,14 +619,57 @@ function countYellingSentences(text: string): number { export function computeUserMessageMetrics(text: string): UserMessageMetrics { const trimmed = text.trim(); if (!trimmed) { - return { chars: 0, words: 0, yellingSentences: 0, profanity: 0, dramaRuns: 0 }; + return { + chars: 0, + words: 0, + yelling: 0, + profanity: 0, + anguish: 0, + negation: 0, + repetition: 0, + blame: 0, + }; } + + const chars = trimmed.length; + const words = countMatches(trimmed, WORD_RE); + + // Behavior signals are computed on a stripped prose body; long / + // well-formatted messages score zero because they are deliberate, not + // emotional outbursts. + const prose = stripStructuredContent(trimmed).trim(); + if (!prose || countNonEmptyLines(prose) >= MAX_PROSE_LINES) { + return { + chars, + words, + yelling: 0, + profanity: 0, + anguish: 0, + negation: 0, + repetition: 0, + blame: 0, + }; + } + + const anguish = + countMatches(prose, DRAMA_RE) + + countMatches(prose, ANGUISH_RE) + + countMatches(prose, DUDE_RE) + + countMatches(prose, ELLIPSIS_RE); + + const negation = countMatches(prose, NEGATION_LEAD_RE) + countMatches(prose, NEGATION_PHRASE_RE); + const repetition = countMatches(prose, REPETITION_RECALL_RE) + countMatches(prose, REPETITION_STILL_RE); + const blame = countMatches(prose, BLAME_YOU_RE) + countMatches(prose, BLAME_STOP_RE); + return { - chars: trimmed.length, - words: countMatches(trimmed, WORD_RE), - yellingSentences: countYellingSentences(trimmed), - profanity: countMatches(trimmed, PROFANITY_RE), - dramaRuns: countMatches(trimmed, DRAMA_RE), + chars, + words, + yelling: countYellingSentences(prose), + profanity: countMatches(prose, PROFANITY_RE), + anguish, + negation, + repetition, + blame, }; } @@ -480,7 +677,10 @@ export function computeUserMessageMetrics(text: string): UserMessageMetrics { export const EMPTY_USER_METRICS: UserMessageMetrics = Object.freeze({ chars: 0, words: 0, - yellingSentences: 0, + yelling: 0, profanity: 0, - dramaRuns: 0, + anguish: 0, + negation: 0, + repetition: 0, + blame: 0, }); diff --git a/packages/stats/test/user-metrics.test.ts b/packages/stats/test/user-metrics.test.ts index 659b805e5..511f62946 100644 --- a/packages/stats/test/user-metrics.test.ts +++ b/packages/stats/test/user-metrics.test.ts @@ -1,85 +1,176 @@ import { describe, expect, it } from "bun:test"; -import { computeUserMessageMetrics } from "../src/user-metrics"; +import { computeUserMessageMetrics, EMPTY_USER_METRICS } from "../src/user-metrics"; describe("computeUserMessageMetrics", () => { it("returns zeros for empty / whitespace-only text", () => { - expect(computeUserMessageMetrics("")).toEqual({ - chars: 0, - words: 0, - yellingSentences: 0, - profanity: 0, - dramaRuns: 0, - }); - expect(computeUserMessageMetrics(" \n\t ")).toEqual({ - chars: 0, - words: 0, - yellingSentences: 0, - profanity: 0, - dramaRuns: 0, - }); + expect(computeUserMessageMetrics("")).toEqual({ ...EMPTY_USER_METRICS }); + expect(computeUserMessageMetrics(" \n\t ")).toEqual({ ...EMPTY_USER_METRICS }); }); it("counts a sentence as yelling when >50% of its letters are uppercase", () => { - // 16 letters, all uppercase → yelling. const m = computeUserMessageMetrics("STOP DOING THAT NOW"); - expect(m.yellingSentences).toBe(1); + expect(m.yelling).toBe(1); }); it("treats mostly-lowercase sentences as not yelling even with embedded CAPS", () => { - // `STOP` and `THAT` are uppercase but the surrounding lowercase keeps the - // per-sentence ratio well under 50%. const m = computeUserMessageMetrics("Hi there, please STOP doing THAT immediately, it is really annoying."); - expect(m.yellingSentences).toBe(0); + expect(m.yelling).toBe(0); }); it("ignores very short uppercase fragments below the letter floor", () => { - // "OK" and "WIP" have fewer than the minimum-letters threshold, so neither - // sentence should register as yelling. - expect(computeUserMessageMetrics("OK").yellingSentences).toBe(0); - expect(computeUserMessageMetrics("WIP.").yellingSentences).toBe(0); + expect(computeUserMessageMetrics("OK").yelling).toBe(0); + expect(computeUserMessageMetrics("WIP.").yelling).toBe(0); }); it("counts multiple yelling sentences separated by terminators", () => { const m = computeUserMessageMetrics("WHY IS THIS BROKEN? FIX IT NOW!! please."); - // "WHY IS THIS BROKEN" and " FIX IT NOW" are both >50% uppercase; the - // trailing " please" sentence is lowercase. - expect(m.yellingSentences).toBe(2); + expect(m.yelling).toBe(2); }); it("does not flag camelCase / acronyms inside otherwise-lowercase prose", () => { const m = computeUserMessageMetrics("call getHTMLParser then exit"); - expect(m.yellingSentences).toBe(0); + expect(m.yelling).toBe(0); }); it("matches profanity case-insensitively at word boundaries only", () => { + // Regression: prior version used a non-raw template literal so `\b` was + // compiled as backspace (U+0008) and the regex matched nothing in real + // prose. Lock the word-boundary contract in. const m = computeUserMessageMetrics("oh FUCK this is bullshit, damn it"); expect(m.profanity).toBe(3); - // `class` shares letters with `ass` but must not match — word boundary required. expect(computeUserMessageMetrics("import classes from module").profanity).toBe(0); }); - it("counts each run of 3+ ! or ? as one drama run", () => { + it("counts quality-dismissal vocabulary as profanity", () => { + const m = computeUserMessageMetrics("this is garbage, useless and horrible work"); + expect(m.profanity).toBe(3); + }); + + it("folds drama runs / elongated interjections / dot trails into `anguish`", () => { const m = computeUserMessageMetrics("why!!! seriously??? omg!?!?!?"); - // "!!!" = 1, "???" = 1, "!?!?!?" = 1 (mixed ≥3 cluster) → 3 - expect(m.dramaRuns).toBe(3); - // Two characters alone do not count as drama. - expect(computeUserMessageMetrics("ok!! sure??").dramaRuns).toBe(0); + expect(m.anguish).toBeGreaterThanOrEqual(3); + expect(computeUserMessageMetrics("ok!! sure??").anguish).toBe(0); }); - it("absorbs shift-key `1` mishits into the surrounding drama run", () => { - // "!!!111" and "!?!?!??111" are both single bursts, not separate hits. - expect(computeUserMessageMetrics("what!!!111").dramaRuns).toBe(1); - expect(computeUserMessageMetrics("are you serious!?!?!??111").dramaRuns).toBe(1); - // Plain digits without a leading `!`/`?` are not drama. - expect(computeUserMessageMetrics("port 8111 please").dramaRuns).toBe(0); + it("absorbs shift-key `1` mishits into a single drama burst", () => { + expect(computeUserMessageMetrics("what!!!111").anguish).toBeGreaterThanOrEqual(1); + expect(computeUserMessageMetrics("are you serious!?!?!??111").anguish).toBeGreaterThanOrEqual(1); + expect(computeUserMessageMetrics("port 8111 please").anguish).toBe(0); }); - it("captures all three signals together with correct chars/words", () => { - const m = computeUserMessageMetrics("WHY IS THIS SO SHITTY???"); - expect(m.yellingSentences).toBe(1); - expect(m.profanity).toBe(1); - expect(m.dramaRuns).toBe(1); - expect(m.words).toBe(5); - expect(m.chars).toBe("WHY IS THIS SO SHITTY???".length); + describe("negation signal", () => { + it("fires on line-leading correction openers", () => { + expect(computeUserMessageMetrics("no this is the renderer").negation).toBe(1); + expect(computeUserMessageMetrics("nope, still wrong").negation).toBe(1); + expect(computeUserMessageMetrics("nah look at this").negation).toBe(1); + expect(computeUserMessageMetrics("wrong file").negation).toBe(1); + expect(computeUserMessageMetrics("nvm got it").negation).toBe(1); + }); + + it("does not fire on words that share a prefix with negation tokens", () => { + // `now`, `nobody`, `north`, `noble`, `normal` all start with `no` but + // are not corrective negation. + expect(computeUserMessageMetrics("now everything works").negation).toBe(0); + expect(computeUserMessageMetrics("nobody knows why").negation).toBe(0); + expect(computeUserMessageMetrics("normal operation resumed").negation).toBe(0); + }); + + it("only anchors at the start of the message - mid-message `no`/`No` lines do not fire", () => { + // Real corrective negation overwhelmingly opens the message. Pasted error + // text and bullet lists trip the old `^...$/m` anchor with FPs like + // "Wrong user name or password" or "No JSDoc warning on X". + expect(computeUserMessageMetrics("i instantly get Finalizing ->\nNo speech detected").negation).toBe(0); + expect(computeUserMessageMetrics("Authentication failed\n\nWrong user name or password").negation).toBe(0); + }); + + it("strips `[Image #N]` placeholders so message-leading negation still fires", () => { + // The TUI inserts `[Image #1]` markers ahead of real user prose; the + // strip pass removes them so anchored-at-start negation still works. + expect(computeUserMessageMetrics("[Image #1] nope still broken").negation).toBe(1); + }); + + it("fires on explicit rejection phrases", () => { + expect(computeUserMessageMetrics("thats not what i wanted").negation).toBe(1); + expect(computeUserMessageMetrics("that's not right").negation).toBe(1); + expect(computeUserMessageMetrics("this is not what i meant at all").negation).toBe(1); + }); + }); + + describe("repetition signal", () => { + it("counts explicit recall verbs", () => { + expect(computeUserMessageMetrics("i meant the other file").repetition).toBe(1); + expect(computeUserMessageMetrics("i told you to skip it").repetition).toBe(1); + expect(computeUserMessageMetrics("i asked you for json not yaml").repetition).toBe(1); + expect(computeUserMessageMetrics("like i said earlier").repetition).toBe(1); + }); + + it("requires `you` after `i asked` to suppress neutral third-party usage", () => { + // In the real corpus, bare `i asked` is overwhelmingly "i asked + // " - committee, experts, weaker LLMs, etc. - + // which is not frustration with us. The `(like|as) i asked` form is + // still allowed because it always refers back to our own ask. + expect(computeUserMessageMetrics("i asked the committee to review").repetition).toBe(0); + expect(computeUserMessageMetrics("so i asked a bunch of experts").repetition).toBe(0); + expect(computeUserMessageMetrics("you're not doing AST rewriting like i asked").repetition).toBe(1); + }); + + it("counts `still` only when paired with a negative / sameness marker", () => { + // Bare `still` would over-fire on neutral usage. + expect(computeUserMessageMetrics("the agent still works fine").repetition).toBe(0); + expect(computeUserMessageMetrics("it still doesnt work").repetition).toBe(1); + expect(computeUserMessageMetrics("still the same issue").repetition).toBe(1); + expect(computeUserMessageMetrics("still failing on darwin").repetition).toBe(1); + }); + }); + + describe("blame signal", () => { + it("fires on accusatory second-person verbs", () => { + expect(computeUserMessageMetrics("you broke the layout").blame).toBe(1); + expect(computeUserMessageMetrics("you didnt update AGENTS").blame).toBe(1); + expect(computeUserMessageMetrics("you missed a callsite").blame).toBe(1); + expect(computeUserMessageMetrics("you forgot to commit").blame).toBe(1); + expect(computeUserMessageMetrics("you keep doing that").blame).toBe(1); + }); + + it("does not fire on bare `you`", () => { + // `you` alone is too generic - dominated by neutral instructions. + expect(computeUserMessageMetrics("can you fix the bug?").blame).toBe(0); + expect(computeUserMessageMetrics("could you also add a test").blame).toBe(0); + }); + + it("only fires on `stop X-ing` at sentence start", () => { + expect(computeUserMessageMetrics("stop touching git").blame).toBe(1); + expect(computeUserMessageMetrics("please stop making yolo changes").blame).toBe(0); // mid-sentence + expect(computeUserMessageMetrics("ok. stop reverting things").blame).toBe(1); + // `nonstop`/`stopping` should not match the imperative pattern. + expect(computeUserMessageMetrics("the loop keeps stopping").blame).toBe(0); + }); + }); + + it("zeros out behavior signals on long structured prompts", () => { + // >= 3 non-empty prose lines after stripping = deliberate prompt, not a tantrum. + const long = [ + "no this is wrong, you broke it, i meant the other one.", + "please undo and try again.", + "acceptance: green tests.", + "thanks!", + ].join("\n"); + const m = computeUserMessageMetrics(long); + expect(m.negation).toBe(0); + expect(m.repetition).toBe(0); + expect(m.blame).toBe(0); + expect(m.anguish).toBe(0); + expect(m.profanity).toBe(0); + expect(m.yelling).toBe(0); + // But char/word counts still reflect the raw text. + expect(m.chars).toBeGreaterThan(0); + expect(m.words).toBeGreaterThan(0); + }); + + it("captures multiple frustration signals on a single short message", () => { + const m = computeUserMessageMetrics("no, you broke it AGAIN. i told you it still doesnt work"); + expect(m.negation).toBe(1); + expect(m.blame).toBe(1); + expect(m.repetition).toBeGreaterThanOrEqual(2); // `i told you` + `still doesnt` }); });