feat(stats): added frustration metrics to stats parser and charting
- Added new frustration and revised behavior metrics across stats types, parser mappings, and UI charts. - Reworked session sync to fan out parsing across a capped worker pool with SyncOptions progress callbacks. - Added worker-based parse messaging and throttled TTY progress rendering in the stats sync command. - Updated behavior scoring to strip structured content, ignore noisy prompts, and fix profanity boundary regressions. - Expanded user message schema and migrations, then repaired assistant model/provider links during sync backfill. - Updated worker-import guidance in AGENTS.md and recorded sync-progress/metrics updates in package changelogs.
This commit is contained in:
@@ -29,6 +29,7 @@ This repo contains multiple packages, but **`packages/coding-agent/`** is the pr
|
||||
- **Class privacy**: use ES `#private` fields; leave externally accessible members bare. **No `private`/`protected`/`public` keyword on fields or methods**, except on **constructor parameter properties** where TypeScript requires it (e.g. `constructor(private readonly session: ToolSession)`).
|
||||
- **Promises**: use `Promise.withResolvers()` instead of `new Promise((resolve, reject) => ...)`.
|
||||
- **Prompts**: never build prompts in code (no inline strings, template literals, or concatenation). Prompts live in static `.md` files; use Handlebars for dynamic content. Import them via `import content from "./prompt.md" with { type: "text" }` — not `readFile`.
|
||||
- **Worker scripts**: reference the entry via `import workerUrl from "./worker.ts" with { type: "file" }` so the path survives bundling. tsgo flags it (TS1192/TS5097) — suppress with `// @ts-expect-error -- Bun file-URL import` directly above the line (see `tab-supervisor.ts`, `context-manager.ts`, `aggregator.ts`).
|
||||
|
||||
## Bun Over Node
|
||||
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Changed the `timeoutMs` execution option to no longer be enforced during worker-based JS runs, so callers must rely on external cancellation signals for time limits
|
||||
|
||||
### Added
|
||||
|
||||
- Added a live single-line sync progress display to the stats command showing current/total sessions while syncing
|
||||
- Added automatic inline JS evaluation fallback when worker creation failed so script execution still works in environments without worker support
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -8,6 +8,58 @@ import { APP_NAME, formatDuration, formatNumber, formatPercent } from "@oh-my-pi
|
||||
import chalk from "chalk";
|
||||
import { openPath } from "../utils/open";
|
||||
|
||||
/**
|
||||
* Single-line TTY progress bar. On a non-TTY stream we just stay quiet -
|
||||
* the final "Synced ..." summary still prints either way.
|
||||
*/
|
||||
function createSyncProgressReporter(): {
|
||||
onProgress: (event: { current: number; total: number; sessionFile: string }) => void;
|
||||
finish: () => void;
|
||||
} {
|
||||
const stream = process.stderr;
|
||||
const isTty = stream.isTTY === true;
|
||||
let lastWidth = 0;
|
||||
let lastRender = 0;
|
||||
return {
|
||||
onProgress(event) {
|
||||
if (!isTty) return;
|
||||
const now = Date.now();
|
||||
// Throttle to ~30 fps and always force a render for the last file.
|
||||
if (event.current < event.total && now - lastRender < 33) return;
|
||||
lastRender = now;
|
||||
const label = chalk.dim(shortenSessionFile(event.sessionFile));
|
||||
const pct = ((event.current / event.total) * 100).toFixed(0).padStart(3, " ");
|
||||
const counter = chalk.cyan(`[${event.current}/${event.total}]`);
|
||||
const line = `${counter} ${pct}% ${label}`;
|
||||
const columns = stream.columns ?? 120;
|
||||
const trimmed = truncateToColumns(line, columns - 1);
|
||||
stream.write(`\r${trimmed.padEnd(lastWidth)}`);
|
||||
lastWidth = trimmed.length;
|
||||
},
|
||||
finish() {
|
||||
if (!isTty || lastWidth === 0) return;
|
||||
stream.write(`\r${" ".repeat(lastWidth)}\r`);
|
||||
lastWidth = 0;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function shortenSessionFile(p: string): string {
|
||||
const marker = "/sessions/";
|
||||
const idx = p.indexOf(marker);
|
||||
return idx >= 0 ? p.slice(idx + marker.length) : p;
|
||||
}
|
||||
|
||||
function truncateToColumns(s: string, max: number): string {
|
||||
if (max <= 0) return "";
|
||||
const width = Bun.stringWidth(s, { countAnsiEscapeCodes: false });
|
||||
if (width <= max) return s;
|
||||
// Cheap right-trim with an ellipsis - we don't need ANSI-aware slicing
|
||||
// because the colored prefix is short and the truncated tail is the
|
||||
// dim filename, where dropping bytes is fine.
|
||||
return `${s.slice(0, Math.max(0, max - 1))}\u2026`;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Types
|
||||
// =============================================================================
|
||||
@@ -74,8 +126,10 @@ export async function runStatsCommand(cmd: StatsCommandArgs): Promise<void> {
|
||||
);
|
||||
|
||||
// Sync session files first
|
||||
console.log("Syncing session files...");
|
||||
const { processed, files } = await syncAllSessions();
|
||||
const progress = createSyncProgressReporter();
|
||||
process.stderr.write("Syncing session files...\n");
|
||||
const { processed, files } = await syncAllSessions({ onProgress: progress.onProgress });
|
||||
progress.finish();
|
||||
const total = await getTotalMessageCount();
|
||||
console.log(`Synced ${processed} new entries from ${files} files (${total} total)\n`);
|
||||
|
||||
|
||||
@@ -503,6 +503,7 @@ function indirectEval(source: string, filename?: string): unknown {
|
||||
const withPragma = filename ? `${source}\n//# sourceURL=${filename}` : source;
|
||||
// Read `eval` via a property access so the call site is *indirect* (global scope),
|
||||
// not direct (this function's lexical scope). The cast erases the lib.dom return type.
|
||||
// biome-ignore lint/security/noGlobalEval: indirect eval is intentional for sandbox execution
|
||||
const geval = globalThis.eval as (src: string) => unknown;
|
||||
return geval(withPragma);
|
||||
}
|
||||
|
||||
@@ -1,6 +1,26 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Breaking Changes
|
||||
|
||||
- Broke backward compatibility of behavior stats fields by replacing `yellingSentences`/`dramaRuns` with `yelling`/`anguish` and adding `negation`, `repetition`, `blame` in query result types and persisted `user_messages` schema
|
||||
|
||||
### Added
|
||||
|
||||
- Added `SyncOptions` to `syncAllSessions` with `onProgress` and `workers` to optionally show per-file sync progress and tune parser concurrency
|
||||
- Added new frustration behavior metrics (`negation`, `repetition`, `blame`) plus a `frustration` aggregate in behavior charts, model tables, and summary cards
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed sync ingestion to parse session files through a worker pool while applying parsed results and database writes on the main thread
|
||||
- Changed behavior analysis to strip code blocks, XML/URLs, quoted lines, and placeholders before scoring and to suppress signals on long structured messages
|
||||
- Changed dashboard metrics labels and totals to the new signal names, including replacing the old three-signal totals with `yelling`, `profanity`, `anguish`, and `frustration`
|
||||
- Changed sync output to print a live terminal progress indicator while processing session files
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed user-message attribution so assistant model/provider links are backfilled during incremental sync instead of being left unknown
|
||||
- Fixed word-boundary regex handling in profanity detection so matching now works as intended in normal prose
|
||||
|
||||
## [14.9.5] - 2026-05-12
|
||||
|
||||
|
||||
@@ -19,67 +19,177 @@ import {
|
||||
insertMessageStats,
|
||||
insertUserMessageStats,
|
||||
setFileOffset,
|
||||
updateUserMessageLinks,
|
||||
} from "./db";
|
||||
import { getSessionEntry, listAllSessionFiles, parseSessionFile } from "./parser";
|
||||
import { getSessionEntry, listAllSessionFiles, type ParseSessionResult } from "./parser";
|
||||
import type { SyncWorkerRequest, SyncWorkerResponse } from "./sync-worker";
|
||||
// `with { type: "file" }` resolves to the worker's absolute path at runtime
|
||||
// (dev) and survives bundling (the asset is copied alongside the build).
|
||||
// tsgo doesn't recognize Bun's file-URL import attribute and would raise
|
||||
// TS1192/TS5097 here; Bun honors it. Same suppression pattern lives in
|
||||
// `tab-supervisor.ts` and `context-manager.ts`.
|
||||
// @ts-expect-error -- Bun file-URL import (see comment above).
|
||||
import syncWorkerUrl from "./sync-worker.ts" with { type: "file" };
|
||||
import type { BehaviorDashboardStats, DashboardStats, MessageStats, RequestDetails } from "./types";
|
||||
|
||||
/**
|
||||
* Sync a single session file to the database.
|
||||
* Only processes new entries since the last sync.
|
||||
* Apply a freshly parsed result to the database. Runs entirely on the
|
||||
* main thread so the single SQLite handle owns every write.
|
||||
*/
|
||||
async function syncSessionFile(sessionFile: string): Promise<number> {
|
||||
// Get file stats
|
||||
let fileStats: fs.Stats;
|
||||
try {
|
||||
fileStats = await fs.promises.stat(sessionFile);
|
||||
} catch {
|
||||
return 0;
|
||||
function applyParseResult(sessionFile: string, lastModified: number, result: ParseSessionResult): number {
|
||||
if (result.stats.length > 0) insertMessageStats(result.stats);
|
||||
if (result.userStats.length > 0) insertUserMessageStats(result.userStats);
|
||||
if (result.userLinks.length > 0) updateUserMessageLinks(result.userLinks);
|
||||
setFileOffset(sessionFile, result.newOffset, lastModified);
|
||||
return result.stats.length + result.userStats.length;
|
||||
}
|
||||
|
||||
/**
|
||||
* Progress event emitted after each session file is fully processed.
|
||||
* `current` is the number of files completed (skipped + parsed),
|
||||
* `total` is the size of the work set. `processed` is the running total
|
||||
* of inserted rows.
|
||||
*/
|
||||
export interface SyncProgress {
|
||||
current: number;
|
||||
total: number;
|
||||
processed: number;
|
||||
sessionFile: string;
|
||||
}
|
||||
|
||||
export interface SyncOptions {
|
||||
/** Called after each file completes. Synchronous; keep it cheap. */
|
||||
onProgress?: (event: SyncProgress) => void;
|
||||
/**
|
||||
* Worker pool size. Defaults to a sensible value derived from the host
|
||||
* (capped to avoid drowning a small machine in workers). Set to `1` to
|
||||
* force serial parsing without spawning workers.
|
||||
*/
|
||||
workers?: number;
|
||||
}
|
||||
|
||||
function defaultWorkerCount(): number {
|
||||
// `navigator.hardwareConcurrency` is the portable answer in Bun; fall
|
||||
// back to a small fixed pool if it's somehow unavailable.
|
||||
const hw = typeof navigator !== "undefined" ? (navigator.hardwareConcurrency ?? 0) : 0;
|
||||
const raw = hw > 0 ? hw : 4;
|
||||
// Cap at 8 - parse is JSON-bound, and SQLite writes serialize on main
|
||||
// thread anyway, so more workers stop helping.
|
||||
return Math.min(8, Math.max(2, Math.floor(raw)));
|
||||
}
|
||||
|
||||
interface WorkerHandle {
|
||||
worker: Worker;
|
||||
busy: boolean;
|
||||
resolve: ((res: ParseSessionResult) => void) | null;
|
||||
reject: ((err: Error) => void) | null;
|
||||
}
|
||||
|
||||
function spawnWorker(): WorkerHandle {
|
||||
const worker = new Worker(syncWorkerUrl, { type: "module" });
|
||||
const handle: WorkerHandle = { worker, busy: false, resolve: null, reject: null };
|
||||
worker.onmessage = (event: MessageEvent<SyncWorkerResponse>) => {
|
||||
const { resolve, reject } = handle;
|
||||
handle.resolve = null;
|
||||
handle.reject = null;
|
||||
handle.busy = false;
|
||||
if (!resolve || !reject) return;
|
||||
if (event.data.ok) resolve(event.data.result);
|
||||
else reject(new Error(event.data.error));
|
||||
};
|
||||
worker.onerror = (event: ErrorEvent) => {
|
||||
const { reject } = handle;
|
||||
handle.resolve = null;
|
||||
handle.reject = null;
|
||||
handle.busy = false;
|
||||
reject?.(event.error instanceof Error ? event.error : new Error(event.message || "worker error"));
|
||||
};
|
||||
return handle;
|
||||
}
|
||||
|
||||
function dispatch(handle: WorkerHandle, request: SyncWorkerRequest): Promise<ParseSessionResult> {
|
||||
if (handle.busy) {
|
||||
return Promise.reject(new Error("worker is busy - this is a bug in the dispatcher"));
|
||||
}
|
||||
|
||||
const lastModified = fileStats.mtimeMs;
|
||||
|
||||
// Check if file has changed since last sync
|
||||
const stored = getFileOffset(sessionFile);
|
||||
if (stored && stored.lastModified >= lastModified) {
|
||||
return 0; // File hasn't changed
|
||||
}
|
||||
|
||||
// Parse file from last offset
|
||||
const fromOffset = stored?.offset ?? 0;
|
||||
const { stats, userStats, newOffset } = await parseSessionFile(sessionFile, fromOffset);
|
||||
|
||||
if (stats.length > 0) {
|
||||
insertMessageStats(stats);
|
||||
}
|
||||
if (userStats.length > 0) {
|
||||
insertUserMessageStats(userStats);
|
||||
}
|
||||
|
||||
// Update offset tracker
|
||||
setFileOffset(sessionFile, newOffset, lastModified);
|
||||
|
||||
return stats.length + userStats.length;
|
||||
const { promise, resolve, reject } = Promise.withResolvers<ParseSessionResult>();
|
||||
handle.busy = true;
|
||||
handle.resolve = resolve;
|
||||
handle.reject = reject;
|
||||
handle.worker.postMessage(request);
|
||||
return promise;
|
||||
}
|
||||
|
||||
/**
|
||||
* Sync all session files to the database.
|
||||
* Returns the number of new entries processed.
|
||||
*
|
||||
* Parsing fans out across a worker pool (one in-flight job per worker)
|
||||
* while DB writes and offset bookkeeping stay on the calling thread so the
|
||||
* single SQLite handle stays uncontended. `onProgress` fires once per
|
||||
* completed file (skipped files included so the bar walks at a steady
|
||||
* rate).
|
||||
*/
|
||||
export async function syncAllSessions(): Promise<{ processed: number; files: number }> {
|
||||
export async function syncAllSessions(opts?: SyncOptions): Promise<{ processed: number; files: number }> {
|
||||
await initDb();
|
||||
|
||||
const files = await listAllSessionFiles();
|
||||
if (files.length === 0) return { processed: 0, files: 0 };
|
||||
|
||||
let totalProcessed = 0;
|
||||
let filesProcessed = 0;
|
||||
let completed = 0;
|
||||
let cursor = 0;
|
||||
|
||||
for (const file of files) {
|
||||
const count = await syncSessionFile(file);
|
||||
if (count > 0) {
|
||||
totalProcessed += count;
|
||||
filesProcessed++;
|
||||
const poolSize = Math.max(1, Math.min(files.length, opts?.workers ?? defaultWorkerCount()));
|
||||
const handles: WorkerHandle[] = [];
|
||||
for (let i = 0; i < poolSize; i++) handles.push(spawnWorker());
|
||||
|
||||
const report = (sessionFile: string) => {
|
||||
completed++;
|
||||
opts?.onProgress?.({
|
||||
current: completed,
|
||||
total: files.length,
|
||||
processed: totalProcessed,
|
||||
sessionFile,
|
||||
});
|
||||
};
|
||||
|
||||
async function drain(handle: WorkerHandle): Promise<void> {
|
||||
while (true) {
|
||||
const idx = cursor++;
|
||||
if (idx >= files.length) return;
|
||||
const sessionFile = files[idx];
|
||||
|
||||
let fileStats: fs.Stats;
|
||||
try {
|
||||
fileStats = await fs.promises.stat(sessionFile);
|
||||
} catch {
|
||||
report(sessionFile);
|
||||
continue;
|
||||
}
|
||||
const lastModified = fileStats.mtimeMs;
|
||||
const stored = getFileOffset(sessionFile);
|
||||
if (stored && stored.lastModified >= lastModified) {
|
||||
report(sessionFile);
|
||||
continue;
|
||||
}
|
||||
|
||||
const fromOffset = stored?.offset ?? 0;
|
||||
const result = await dispatch(handle, { sessionFile, fromOffset });
|
||||
const inserted = applyParseResult(sessionFile, lastModified, result);
|
||||
if (inserted > 0) {
|
||||
totalProcessed += inserted;
|
||||
filesProcessed++;
|
||||
}
|
||||
report(sessionFile);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
await Promise.all(handles.map(drain));
|
||||
} finally {
|
||||
for (const handle of handles) handle.worker.terminate();
|
||||
}
|
||||
|
||||
return { processed: totalProcessed, files: filesProcessed };
|
||||
}
|
||||
|
||||
|
||||
@@ -51,10 +51,14 @@ const CHART_THEMES = {
|
||||
} as const;
|
||||
|
||||
const METRIC_OPTIONS = [
|
||||
{ value: "yellingSentences", label: "Yelling" },
|
||||
{ value: "yelling", label: "Yelling" },
|
||||
{ value: "profanity", label: "Profanity" },
|
||||
{ value: "dramaRuns", label: "Drama (!!! / ???)" },
|
||||
{ value: "total", label: "All three combined" },
|
||||
{ value: "anguish", label: "Anguish (!!!, nooo, dude, ..)" },
|
||||
{ value: "negation", label: "Negation (no/nope/wrong)" },
|
||||
{ value: "repetition", label: "Repetition (i meant, still doesnt)" },
|
||||
{ value: "blame", label: "Blame (you didnt, stop X-ing)" },
|
||||
{ value: "frustration", label: "Frustration (neg + rep + blame)" },
|
||||
{ value: "total", label: "All signals combined" },
|
||||
] as const;
|
||||
type Metric = (typeof METRIC_OPTIONS)[number]["value"];
|
||||
|
||||
@@ -70,7 +74,10 @@ interface BehaviorChartProps {
|
||||
}
|
||||
|
||||
function pointHits(point: BehaviorTimeSeriesPoint, metric: Metric): number {
|
||||
if (metric === "total") return point.yellingSentences + point.profanity + point.dramaRuns;
|
||||
if (metric === "frustration") return point.negation + point.repetition + point.blame;
|
||||
if (metric === "total") {
|
||||
return point.yelling + point.profanity + point.anguish + point.negation + point.repetition + point.blame;
|
||||
}
|
||||
return point[metric];
|
||||
}
|
||||
|
||||
|
||||
@@ -30,7 +30,8 @@ const MODEL_COLORS = [
|
||||
const SERIES_COLORS = {
|
||||
yelling: "#fbbf24", // amber
|
||||
profanity: "#f87171", // red
|
||||
drama: "#a78bfa", // violet
|
||||
anguish: "#a78bfa", // violet
|
||||
frustration: "#22d3ee", // cyan - new semantic signals
|
||||
} as const;
|
||||
|
||||
const CHART_THEMES = {
|
||||
@@ -65,7 +66,8 @@ interface DailyPoint {
|
||||
timestamp: number;
|
||||
yelling: number;
|
||||
profanity: number;
|
||||
drama: number;
|
||||
anguish: number;
|
||||
frustration: number;
|
||||
total: number;
|
||||
}
|
||||
|
||||
@@ -73,7 +75,7 @@ interface ModelTrendSeries {
|
||||
data: DailyPoint[];
|
||||
}
|
||||
|
||||
const GRID_TEMPLATE = "2fr 0.9fr 0.9fr 0.9fr 0.9fr 0.9fr 140px 40px";
|
||||
const GRID_TEMPLATE = "2fr 0.9fr 0.8fr 0.8fr 0.8fr 0.9fr 0.8fr 140px 40px";
|
||||
|
||||
function formatInt(value: number): string {
|
||||
return value.toLocaleString();
|
||||
@@ -81,7 +83,13 @@ function formatInt(value: number): string {
|
||||
|
||||
function totalHitRate(model: BehaviorModelStats): number {
|
||||
if (model.totalMessages === 0) return 0;
|
||||
const hits = model.totalYellingSentences + model.totalProfanity + model.totalDramaRuns;
|
||||
const hits =
|
||||
model.totalYelling +
|
||||
model.totalProfanity +
|
||||
model.totalAnguish +
|
||||
model.totalNegation +
|
||||
model.totalRepetition +
|
||||
model.totalBlame;
|
||||
return hits / model.totalMessages;
|
||||
}
|
||||
|
||||
@@ -128,7 +136,8 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
|
||||
<div className="text-right">Messages</div>
|
||||
<div className="text-right">CAPS %</div>
|
||||
<div className="text-right">Profanity %</div>
|
||||
<div className="text-right">Drama %</div>
|
||||
<div className="text-right">Anguish %</div>
|
||||
<div className="text-right">Frustration %</div>
|
||||
<div className="text-right">Hits %</div>
|
||||
<div className="text-center">Trend</div>
|
||||
<div />
|
||||
@@ -140,7 +149,8 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
|
||||
const trend = trendByKey.get(key)?.data ?? [];
|
||||
const trendColor = MODEL_COLORS[index % MODEL_COLORS.length];
|
||||
const isExpanded = expandedKey === key;
|
||||
const totalHits = model.totalYellingSentences + model.totalProfanity + model.totalDramaRuns;
|
||||
const totalFrustration = model.totalNegation + model.totalRepetition + model.totalBlame;
|
||||
const totalHits = model.totalYelling + model.totalProfanity + model.totalAnguish + totalFrustration;
|
||||
|
||||
return (
|
||||
<div key={key} className="border-t border-[var(--border-subtle)]">
|
||||
@@ -158,13 +168,16 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
|
||||
{formatInt(model.totalMessages)}
|
||||
</div>
|
||||
<div className="text-right text-[var(--text-secondary)] font-mono text-sm">
|
||||
{formatRate(model.totalYellingSentences, model.totalMessages)}
|
||||
{formatRate(model.totalYelling, model.totalMessages)}
|
||||
</div>
|
||||
<div className="text-right text-[var(--text-secondary)] font-mono text-sm">
|
||||
{formatRate(model.totalProfanity, model.totalMessages)}
|
||||
</div>
|
||||
<div className="text-right text-[var(--text-secondary)] font-mono text-sm">
|
||||
{formatRate(model.totalDramaRuns, model.totalMessages)}
|
||||
{formatRate(model.totalAnguish, model.totalMessages)}
|
||||
</div>
|
||||
<div className="text-right text-[var(--text-secondary)] font-mono text-sm">
|
||||
{formatRate(totalFrustration, model.totalMessages)}
|
||||
</div>
|
||||
<div className="text-right text-[var(--text-secondary)] font-mono text-sm">
|
||||
{formatRate(totalHits, model.totalMessages)}
|
||||
@@ -188,7 +201,7 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
|
||||
<div className="space-y-4 text-sm">
|
||||
<DetailRow
|
||||
label="Yelling (CAPS)"
|
||||
total={model.totalYellingSentences}
|
||||
total={model.totalYelling}
|
||||
messages={model.totalMessages}
|
||||
valueClass="text-[var(--accent-amber,#fbbf24)]"
|
||||
/>
|
||||
@@ -199,11 +212,29 @@ export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTa
|
||||
valueClass="text-[var(--accent-red,#f87171)]"
|
||||
/>
|
||||
<DetailRow
|
||||
label="Drama (!!! / ???)"
|
||||
total={model.totalDramaRuns}
|
||||
label="Anguish (!!!, nooo, dude, ..)"
|
||||
total={model.totalAnguish}
|
||||
messages={model.totalMessages}
|
||||
valueClass="text-[var(--accent-violet,#a78bfa)]"
|
||||
/>
|
||||
<DetailRow
|
||||
label="Negation (no/nope/wrong)"
|
||||
total={model.totalNegation}
|
||||
messages={model.totalMessages}
|
||||
valueClass="text-[var(--accent-cyan,#22d3ee)]"
|
||||
/>
|
||||
<DetailRow
|
||||
label="Repetition (i meant, still doesnt)"
|
||||
total={model.totalRepetition}
|
||||
messages={model.totalMessages}
|
||||
valueClass="text-[var(--accent-cyan,#22d3ee)]"
|
||||
/>
|
||||
<DetailRow
|
||||
label="Blame (you didnt, stop X-ing)"
|
||||
total={model.totalBlame}
|
||||
messages={model.totalMessages}
|
||||
valueClass="text-[var(--accent-cyan,#22d3ee)]"
|
||||
/>
|
||||
<DetailRow
|
||||
label="Avg chars / msg"
|
||||
total={model.totalChars}
|
||||
@@ -322,9 +353,18 @@ function BreakdownChart({ data, chartTheme }: { data: DailyPoint[]; chartTheme:
|
||||
borderWidth: 2,
|
||||
},
|
||||
{
|
||||
label: "Drama",
|
||||
data: data.map(d => d.drama),
|
||||
borderColor: SERIES_COLORS.drama,
|
||||
label: "Anguish",
|
||||
data: data.map(d => d.anguish),
|
||||
borderColor: SERIES_COLORS.anguish,
|
||||
backgroundColor: "transparent",
|
||||
tension: 0.4,
|
||||
pointRadius: 0,
|
||||
borderWidth: 2,
|
||||
},
|
||||
{
|
||||
label: "Frustration",
|
||||
data: data.map(d => d.frustration),
|
||||
borderColor: SERIES_COLORS.frustration,
|
||||
backgroundColor: "transparent",
|
||||
tension: 0.4,
|
||||
pointRadius: 0,
|
||||
@@ -394,13 +434,15 @@ function buildTrendLookup(points: BehaviorTimeSeriesPoint[]): Map<string, ModelT
|
||||
timestamp: point.timestamp,
|
||||
yelling: 0,
|
||||
profanity: 0,
|
||||
drama: 0,
|
||||
anguish: 0,
|
||||
frustration: 0,
|
||||
total: 0,
|
||||
};
|
||||
existing.yelling += point.yellingSentences;
|
||||
existing.yelling += point.yelling;
|
||||
existing.profanity += point.profanity;
|
||||
existing.drama += point.dramaRuns;
|
||||
existing.total = existing.yelling + existing.profanity + existing.drama;
|
||||
existing.anguish += point.anguish;
|
||||
existing.frustration += point.negation + point.repetition + point.blame;
|
||||
existing.total = existing.yelling + existing.profanity + existing.anguish + existing.frustration;
|
||||
dayMap.set(point.timestamp, existing);
|
||||
}
|
||||
|
||||
@@ -412,7 +454,8 @@ function buildTrendLookup(points: BehaviorTimeSeriesPoint[]): Map<string, ModelT
|
||||
timestamp: ts,
|
||||
yelling: 0,
|
||||
profanity: 0,
|
||||
drama: 0,
|
||||
anguish: 0,
|
||||
frustration: 0,
|
||||
total: 0,
|
||||
},
|
||||
);
|
||||
|
||||
@@ -10,14 +10,26 @@ function formatInt(value: number): string {
|
||||
return value.toLocaleString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-message rate for a signal. Uses 2 decimals so a 0.01-hits-per-msg model
|
||||
* still distinguishes from a true zero, and never shows `NaN` or `Infinity`
|
||||
* when there are no messages.
|
||||
*/
|
||||
function perMsg(total: number, messages: number): string | undefined {
|
||||
if (messages <= 0) return undefined;
|
||||
return `${(total / messages).toFixed(2)} / msg`;
|
||||
}
|
||||
|
||||
export function BehaviorSummary({ overall, behaviorSeries }: BehaviorSummaryProps) {
|
||||
// Top "ranted-at" model: model that absorbed the most caps + profanity + drama.
|
||||
// Top "ranted-at" model: model that absorbed the most caps + profanity +
|
||||
// anguish + frustration (negation/repetition/blame).
|
||||
const topModel = useMemo(() => {
|
||||
const totals = new Map<string, { model: string; provider: string; score: number }>();
|
||||
for (const point of behaviorSeries) {
|
||||
const key = `${point.model}::${point.provider}`;
|
||||
const existing = totals.get(key);
|
||||
const score = point.yellingSentences + point.profanity + point.dramaRuns;
|
||||
const score =
|
||||
point.yelling + point.profanity + point.anguish + point.negation + point.repetition + point.blame;
|
||||
if (existing) {
|
||||
existing.score += score;
|
||||
} else {
|
||||
@@ -31,36 +43,44 @@ export function BehaviorSummary({ overall, behaviorSeries }: BehaviorSummaryProp
|
||||
return best;
|
||||
}, [behaviorSeries]);
|
||||
|
||||
const capsPerMsg = overall.totalMessages > 0 ? overall.totalYellingSentences / overall.totalMessages : 0;
|
||||
const totalFrustration = overall.totalNegation + overall.totalRepetition + overall.totalBlame;
|
||||
const messages = overall.totalMessages;
|
||||
|
||||
const cards: Array<{ label: string; value: string; sub?: string }> = [
|
||||
{
|
||||
label: "Messages",
|
||||
value: formatInt(overall.totalMessages),
|
||||
sub: messages > 0 ? "in selected range" : undefined,
|
||||
},
|
||||
{
|
||||
label: "Yelling",
|
||||
value: formatInt(overall.totalYellingSentences),
|
||||
sub: overall.totalMessages > 0 ? `${capsPerMsg.toFixed(2)} / msg` : undefined,
|
||||
value: formatInt(overall.totalYelling),
|
||||
sub: perMsg(overall.totalYelling, messages),
|
||||
},
|
||||
{
|
||||
label: "Profanity hits",
|
||||
value: formatInt(overall.totalProfanity),
|
||||
sub: perMsg(overall.totalProfanity, messages),
|
||||
},
|
||||
{
|
||||
label: "Drama runs",
|
||||
value: formatInt(overall.totalDramaRuns),
|
||||
sub: "!!! / ???",
|
||||
label: "Anguish",
|
||||
value: formatInt(overall.totalAnguish),
|
||||
sub: perMsg(overall.totalAnguish, messages),
|
||||
},
|
||||
{
|
||||
label: "Frustration",
|
||||
value: formatInt(totalFrustration),
|
||||
sub: perMsg(totalFrustration, messages),
|
||||
},
|
||||
{
|
||||
label: "Most yelled-at",
|
||||
value: topModel?.model ?? "—",
|
||||
value: topModel?.model ?? "\u2014",
|
||||
sub: topModel ? `${formatInt(topModel.score)} hits` : undefined,
|
||||
},
|
||||
];
|
||||
|
||||
return (
|
||||
<div className="grid grid-cols-2 sm:grid-cols-5 gap-4">
|
||||
<div className="grid grid-cols-2 sm:grid-cols-6 gap-4">
|
||||
{cards.map(card => (
|
||||
<div key={card.label} className="surface px-4 py-3">
|
||||
<p className="text-xs text-[var(--text-muted)] mb-1">{card.label}</p>
|
||||
|
||||
@@ -135,17 +135,23 @@ export interface BehaviorTimeSeriesPoint {
|
||||
model: string;
|
||||
provider: string;
|
||||
messages: number;
|
||||
yellingSentences: number;
|
||||
yelling: number;
|
||||
profanity: number;
|
||||
dramaRuns: number;
|
||||
anguish: number;
|
||||
negation: number;
|
||||
repetition: number;
|
||||
blame: number;
|
||||
chars: number;
|
||||
}
|
||||
|
||||
export interface BehaviorOverallStats {
|
||||
totalMessages: number;
|
||||
totalYellingSentences: number;
|
||||
totalYelling: number;
|
||||
totalProfanity: number;
|
||||
totalDramaRuns: number;
|
||||
totalAnguish: number;
|
||||
totalNegation: number;
|
||||
totalRepetition: number;
|
||||
totalBlame: number;
|
||||
totalChars: number;
|
||||
firstTimestamp: number;
|
||||
lastTimestamp: number;
|
||||
@@ -155,9 +161,12 @@ export interface BehaviorModelStats {
|
||||
model: string;
|
||||
provider: string;
|
||||
totalMessages: number;
|
||||
totalYellingSentences: number;
|
||||
totalYelling: number;
|
||||
totalProfanity: number;
|
||||
totalDramaRuns: number;
|
||||
totalAnguish: number;
|
||||
totalNegation: number;
|
||||
totalRepetition: number;
|
||||
totalBlame: number;
|
||||
totalChars: number;
|
||||
lastTimestamp: number;
|
||||
}
|
||||
|
||||
+151
-38
@@ -14,6 +14,7 @@ import type {
|
||||
ModelStats,
|
||||
ModelTimeSeriesPoint,
|
||||
TimeSeriesPoint,
|
||||
UserMessageLink,
|
||||
UserMessageStats,
|
||||
} from "./types";
|
||||
|
||||
@@ -98,9 +99,12 @@ export async function initDb(): Promise<Database> {
|
||||
provider TEXT,
|
||||
chars INTEGER NOT NULL,
|
||||
words INTEGER NOT NULL,
|
||||
yelling_sentences INTEGER NOT NULL,
|
||||
yelling INTEGER NOT NULL,
|
||||
profanity INTEGER NOT NULL,
|
||||
drama_runs INTEGER NOT NULL,
|
||||
anguish INTEGER NOT NULL,
|
||||
negation INTEGER NOT NULL DEFAULT 0,
|
||||
repetition INTEGER NOT NULL DEFAULT 0,
|
||||
blame INTEGER NOT NULL DEFAULT 0,
|
||||
UNIQUE(session_file, entry_id)
|
||||
);
|
||||
|
||||
@@ -118,16 +122,30 @@ export async function initDb(): Promise<Database> {
|
||||
db.exec("ALTER TABLE messages ADD COLUMN premium_requests REAL NOT NULL DEFAULT 0");
|
||||
}
|
||||
db.exec("UPDATE messages SET premium_requests = 0 WHERE premium_requests IS NULL");
|
||||
// Bumping the metric definition (yelling sentences vs caps words) invalidates
|
||||
// previously-ingested rows. If the legacy column is present we drop the table
|
||||
// outright; the `IF NOT EXISTS` create above already gave us the new schema
|
||||
// in parallel, but we want a clean wipe + re-ingest. The accompanying
|
||||
// `backfillUserMessages` bump (v2) clears `file_offsets` so the next sync
|
||||
// re-parses every session.
|
||||
// Each behavior-metric bump invalidates previously-ingested rows. We detect
|
||||
// the stale schema by column name and drop the table; `IF NOT EXISTS` above
|
||||
// already produced the new schema, but we want a clean wipe + re-ingest.
|
||||
// `backfillUserMessages` then clears `file_offsets` so the next sync
|
||||
// re-parses every session under the current metric definitions.
|
||||
// v1 -> v2: yelling sentences replace `caps_words`.
|
||||
// v2 -> v3: `drama_runs` folded into a single `anguish` signal that
|
||||
// also captures elongated interjections, `dude`, and dot runs,
|
||||
// gated on a stripped prose-line budget.
|
||||
// v3 -> v4: added `negation`, `repetition`, `blame` frustration signals
|
||||
// plus profanity dictionary expansion + word-boundary fix.
|
||||
// v4 -> v5: column `yelling_sentences` renamed to `yelling` to match
|
||||
// the other single-word signal columns.
|
||||
const userMessageColumns = db.prepare("PRAGMA table_info(user_messages)").all() as {
|
||||
name: string;
|
||||
}[];
|
||||
if (userMessageColumns.some(column => column.name === "caps_words")) {
|
||||
const hasStaleColumn =
|
||||
userMessageColumns.length > 0 &&
|
||||
(userMessageColumns.some(column => column.name === "caps_words") ||
|
||||
userMessageColumns.some(column => column.name === "drama_runs") ||
|
||||
userMessageColumns.some(column => column.name === "yelling_sentences"));
|
||||
const hasV4Columns = userMessageColumns.some(column => column.name === "negation");
|
||||
const hasOldUserMessages = userMessageColumns.length > 0;
|
||||
if (hasStaleColumn || (hasOldUserMessages && !hasV4Columns)) {
|
||||
db.exec("DROP TABLE user_messages");
|
||||
db.exec(`
|
||||
CREATE TABLE user_messages (
|
||||
@@ -140,9 +158,12 @@ export async function initDb(): Promise<Database> {
|
||||
provider TEXT,
|
||||
chars INTEGER NOT NULL,
|
||||
words INTEGER NOT NULL,
|
||||
yelling_sentences INTEGER NOT NULL,
|
||||
yelling INTEGER NOT NULL,
|
||||
profanity INTEGER NOT NULL,
|
||||
drama_runs INTEGER NOT NULL,
|
||||
anguish INTEGER NOT NULL,
|
||||
negation INTEGER NOT NULL DEFAULT 0,
|
||||
repetition INTEGER NOT NULL DEFAULT 0,
|
||||
blame INTEGER NOT NULL DEFAULT 0,
|
||||
UNIQUE(session_file, entry_id)
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_user_messages_timestamp ON user_messages(timestamp);
|
||||
@@ -150,6 +171,7 @@ export async function initDb(): Promise<Database> {
|
||||
`);
|
||||
}
|
||||
backfillUserMessages(db);
|
||||
repairUserMessageLinks(db);
|
||||
backfillMissingCatalogCosts(db);
|
||||
return db;
|
||||
}
|
||||
@@ -704,11 +726,19 @@ export function getCostTimeSeries(days = 90, cutoff?: number | null): CostTimeSe
|
||||
* - v1: initial introduction of `user_messages`.
|
||||
* - v2: yelling-sentence metric replaces caps-word counts; existing rows are
|
||||
* computed under the old definition and must be discarded.
|
||||
* - v3: drama runs collapsed into `anguish` (drama + elongated interjections
|
||||
* + `dude` + dot runs), scored on a stripped prose body and gated on
|
||||
* line count. Existing rows used the narrower definition.
|
||||
* - v4: added `negation` / `repetition` / `blame` signals and fixed a
|
||||
* latent word-boundary bug in the profanity / anguish regexes that had
|
||||
* left those metrics matching nothing in real prose.
|
||||
* - v5: renamed `yelling_sentences` column to `yelling` to match the other
|
||||
* single-word signal columns (profanity, anguish, negation, ...).
|
||||
*
|
||||
* Existing `messages` rows are unaffected — `INSERT OR IGNORE` keeps them.
|
||||
* Existing `messages` rows are unaffected - `INSERT OR IGNORE` keeps them.
|
||||
*/
|
||||
function backfillUserMessages(database: Database): void {
|
||||
const row = database.prepare("SELECT value FROM meta WHERE key = 'user_messages_v2'").get() as
|
||||
const row = database.prepare("SELECT value FROM meta WHERE key = 'user_messages_v5'").get() as
|
||||
| { value: string }
|
||||
| undefined;
|
||||
if (row) return;
|
||||
@@ -717,7 +747,27 @@ function backfillUserMessages(database: Database): void {
|
||||
database.exec("DELETE FROM file_offsets");
|
||||
database
|
||||
.prepare("INSERT OR REPLACE INTO meta (key, value) VALUES (?, ?)")
|
||||
.run("user_messages_v2", String(Date.now()));
|
||||
.run("user_messages_v5", String(Date.now()));
|
||||
}
|
||||
|
||||
/**
|
||||
* One-shot wipe of `file_offsets` to force `parseSessionFile` to re-parse
|
||||
* every session from byte zero. We don't touch `user_messages`; the parser
|
||||
* now emits a `UserMessageLink` for every assistant->parent pair, and the
|
||||
* guarded `updateUserMessageLinks` UPDATE fixes any row whose `model` was
|
||||
* left NULL by the old in-pass-only linking logic. Idempotent: gated by a
|
||||
* sentinel row in `meta`.
|
||||
*/
|
||||
function repairUserMessageLinks(database: Database): void {
|
||||
const row = database.prepare("SELECT value FROM meta WHERE key = 'user_message_links_v1'").get() as
|
||||
| { value: string }
|
||||
| undefined;
|
||||
if (row) return;
|
||||
|
||||
database.exec("DELETE FROM file_offsets");
|
||||
database
|
||||
.prepare("INSERT OR REPLACE INTO meta (key, value) VALUES (?, ?)")
|
||||
.run("user_message_links_v1", String(Date.now()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -729,8 +779,9 @@ export function insertUserMessageStats(stats: UserMessageStats[]): number {
|
||||
const stmt = db.prepare(`
|
||||
INSERT OR IGNORE INTO user_messages (
|
||||
session_file, entry_id, folder, timestamp, model, provider,
|
||||
chars, words, yelling_sentences, profanity, drama_runs
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
chars, words, yelling, profanity, anguish,
|
||||
negation, repetition, blame
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`);
|
||||
|
||||
let inserted = 0;
|
||||
@@ -745,9 +796,12 @@ export function insertUserMessageStats(stats: UserMessageStats[]): number {
|
||||
s.provider,
|
||||
s.chars,
|
||||
s.words,
|
||||
s.yellingSentences,
|
||||
s.yelling,
|
||||
s.profanity,
|
||||
s.dramaRuns,
|
||||
s.anguish,
|
||||
s.negation,
|
||||
s.repetition,
|
||||
s.blame,
|
||||
);
|
||||
if (result.changes > 0) inserted++;
|
||||
}
|
||||
@@ -756,6 +810,35 @@ export function insertUserMessageStats(stats: UserMessageStats[]): number {
|
||||
return inserted;
|
||||
}
|
||||
|
||||
/**
|
||||
* Backfill the responding `model`/`provider` on user-message rows that were
|
||||
* persisted before their assistant reply was parsed (a side effect of
|
||||
* incremental `fromOffset` syncing: the `userByEntryId` map in
|
||||
* `parseSessionFile` only spans a single pass). Each row is updated at most
|
||||
* once because the `model IS NULL` guard short-circuits subsequent passes.
|
||||
*
|
||||
* Returns the number of rows actually updated.
|
||||
*/
|
||||
export function updateUserMessageLinks(links: UserMessageLink[]): number {
|
||||
if (!db || links.length === 0) return 0;
|
||||
|
||||
const stmt = db.prepare(`
|
||||
UPDATE user_messages
|
||||
SET model = ?, provider = ?
|
||||
WHERE session_file = ? AND entry_id = ? AND model IS NULL
|
||||
`);
|
||||
|
||||
let updated = 0;
|
||||
const apply = db.transaction(() => {
|
||||
for (const link of links) {
|
||||
const result = stmt.run(link.model, link.provider, link.sessionFile, link.entryId);
|
||||
if (result.changes > 0) updated++;
|
||||
}
|
||||
});
|
||||
apply();
|
||||
return updated;
|
||||
}
|
||||
|
||||
const UNKNOWN_MODEL = "unknown";
|
||||
|
||||
interface BehaviorSeriesRow {
|
||||
@@ -763,9 +846,12 @@ interface BehaviorSeriesRow {
|
||||
model: string;
|
||||
provider: string;
|
||||
messages: number;
|
||||
yelling_sentences: number | null;
|
||||
yelling: number | null;
|
||||
profanity: number | null;
|
||||
drama_runs: number | null;
|
||||
anguish: number | null;
|
||||
negation: number | null;
|
||||
repetition: number | null;
|
||||
blame: number | null;
|
||||
chars: number | null;
|
||||
}
|
||||
|
||||
@@ -781,9 +867,12 @@ export function getBehaviorTimeSeries(cutoff?: number | null): BehaviorTimeSerie
|
||||
COALESCE(model, ?) as model,
|
||||
COALESCE(provider, ?) as provider,
|
||||
COUNT(*) as messages,
|
||||
SUM(yelling_sentences) as yelling_sentences,
|
||||
SUM(yelling) as yelling,
|
||||
SUM(profanity) as profanity,
|
||||
SUM(drama_runs) as drama_runs,
|
||||
SUM(anguish) as anguish,
|
||||
SUM(negation) as negation,
|
||||
SUM(repetition) as repetition,
|
||||
SUM(blame) as blame,
|
||||
SUM(chars) as chars
|
||||
FROM user_messages
|
||||
${hasCutoff ? "WHERE timestamp >= ?" : ""}
|
||||
@@ -798,18 +887,24 @@ export function getBehaviorTimeSeries(cutoff?: number | null): BehaviorTimeSerie
|
||||
model: row.model,
|
||||
provider: row.provider,
|
||||
messages: row.messages,
|
||||
yellingSentences: row.yelling_sentences ?? 0,
|
||||
yelling: row.yelling ?? 0,
|
||||
profanity: row.profanity ?? 0,
|
||||
dramaRuns: row.drama_runs ?? 0,
|
||||
anguish: row.anguish ?? 0,
|
||||
negation: row.negation ?? 0,
|
||||
repetition: row.repetition ?? 0,
|
||||
blame: row.blame ?? 0,
|
||||
chars: row.chars ?? 0,
|
||||
}));
|
||||
}
|
||||
|
||||
interface BehaviorOverallRow {
|
||||
total_messages: number;
|
||||
total_yelling_sentences: number | null;
|
||||
total_yelling: number | null;
|
||||
total_profanity: number | null;
|
||||
total_drama_runs: number | null;
|
||||
total_anguish: number | null;
|
||||
total_negation: number | null;
|
||||
total_repetition: number | null;
|
||||
total_blame: number | null;
|
||||
total_chars: number | null;
|
||||
first_timestamp: number | null;
|
||||
last_timestamp: number | null;
|
||||
@@ -821,9 +916,12 @@ interface BehaviorOverallRow {
|
||||
export function getBehaviorOverall(cutoff?: number | null): BehaviorOverallStats {
|
||||
const empty: BehaviorOverallStats = {
|
||||
totalMessages: 0,
|
||||
totalYellingSentences: 0,
|
||||
totalYelling: 0,
|
||||
totalProfanity: 0,
|
||||
totalDramaRuns: 0,
|
||||
totalAnguish: 0,
|
||||
totalNegation: 0,
|
||||
totalRepetition: 0,
|
||||
totalBlame: 0,
|
||||
totalChars: 0,
|
||||
firstTimestamp: 0,
|
||||
lastTimestamp: 0,
|
||||
@@ -833,9 +931,12 @@ export function getBehaviorOverall(cutoff?: number | null): BehaviorOverallStats
|
||||
const stmt = db.prepare(`
|
||||
SELECT
|
||||
COUNT(*) as total_messages,
|
||||
SUM(yelling_sentences) as total_yelling_sentences,
|
||||
SUM(yelling) as total_yelling,
|
||||
SUM(profanity) as total_profanity,
|
||||
SUM(drama_runs) as total_drama_runs,
|
||||
SUM(anguish) as total_anguish,
|
||||
SUM(negation) as total_negation,
|
||||
SUM(repetition) as total_repetition,
|
||||
SUM(blame) as total_blame,
|
||||
SUM(chars) as total_chars,
|
||||
MIN(timestamp) as first_timestamp,
|
||||
MAX(timestamp) as last_timestamp
|
||||
@@ -846,9 +947,12 @@ export function getBehaviorOverall(cutoff?: number | null): BehaviorOverallStats
|
||||
if (!row?.total_messages) return empty;
|
||||
return {
|
||||
totalMessages: row.total_messages,
|
||||
totalYellingSentences: row.total_yelling_sentences ?? 0,
|
||||
totalYelling: row.total_yelling ?? 0,
|
||||
totalProfanity: row.total_profanity ?? 0,
|
||||
totalDramaRuns: row.total_drama_runs ?? 0,
|
||||
totalAnguish: row.total_anguish ?? 0,
|
||||
totalNegation: row.total_negation ?? 0,
|
||||
totalRepetition: row.total_repetition ?? 0,
|
||||
totalBlame: row.total_blame ?? 0,
|
||||
totalChars: row.total_chars ?? 0,
|
||||
firstTimestamp: row.first_timestamp ?? 0,
|
||||
lastTimestamp: row.last_timestamp ?? 0,
|
||||
@@ -859,9 +963,12 @@ interface BehaviorByModelRow {
|
||||
model: string;
|
||||
provider: string;
|
||||
total_messages: number;
|
||||
total_yelling_sentences: number | null;
|
||||
total_yelling: number | null;
|
||||
total_profanity: number | null;
|
||||
total_drama_runs: number | null;
|
||||
total_anguish: number | null;
|
||||
total_negation: number | null;
|
||||
total_repetition: number | null;
|
||||
total_blame: number | null;
|
||||
total_chars: number | null;
|
||||
last_timestamp: number | null;
|
||||
}
|
||||
@@ -878,9 +985,12 @@ export function getBehaviorByModel(cutoff?: number | null): BehaviorModelStats[]
|
||||
COALESCE(model, ?) as model,
|
||||
COALESCE(provider, ?) as provider,
|
||||
COUNT(*) as total_messages,
|
||||
SUM(yelling_sentences) as total_yelling_sentences,
|
||||
SUM(yelling) as total_yelling,
|
||||
SUM(profanity) as total_profanity,
|
||||
SUM(drama_runs) as total_drama_runs,
|
||||
SUM(anguish) as total_anguish,
|
||||
SUM(negation) as total_negation,
|
||||
SUM(repetition) as total_repetition,
|
||||
SUM(blame) as total_blame,
|
||||
SUM(chars) as total_chars,
|
||||
MAX(timestamp) as last_timestamp
|
||||
FROM user_messages
|
||||
@@ -895,9 +1005,12 @@ export function getBehaviorByModel(cutoff?: number | null): BehaviorModelStats[]
|
||||
model: row.model,
|
||||
provider: row.provider,
|
||||
totalMessages: row.total_messages,
|
||||
totalYellingSentences: row.total_yelling_sentences ?? 0,
|
||||
totalYelling: row.total_yelling ?? 0,
|
||||
totalProfanity: row.total_profanity ?? 0,
|
||||
totalDramaRuns: row.total_drama_runs ?? 0,
|
||||
totalAnguish: row.total_anguish ?? 0,
|
||||
totalNegation: row.total_negation ?? 0,
|
||||
totalRepetition: row.total_repetition ?? 0,
|
||||
totalBlame: row.total_blame ?? 0,
|
||||
totalChars: row.total_chars ?? 0,
|
||||
lastTimestamp: row.last_timestamp ?? 0,
|
||||
}));
|
||||
|
||||
@@ -6,7 +6,13 @@ import { getDashboardStats, getTotalMessageCount, syncAllSessions } from "./aggr
|
||||
import { closeDb } from "./db";
|
||||
import { startServer } from "./server";
|
||||
|
||||
export { getDashboardStats, getTotalMessageCount, syncAllSessions } from "./aggregator";
|
||||
export {
|
||||
getDashboardStats,
|
||||
getTotalMessageCount,
|
||||
type SyncOptions,
|
||||
type SyncProgress,
|
||||
syncAllSessions,
|
||||
} from "./aggregator";
|
||||
export { closeDb } from "./db";
|
||||
export { startServer } from "./server";
|
||||
export type {
|
||||
@@ -112,8 +118,28 @@ Examples:
|
||||
|
||||
try {
|
||||
// Sync first
|
||||
console.log("Syncing session files...");
|
||||
const { processed, files } = await syncAllSessions();
|
||||
const tty = process.stderr.isTTY === true;
|
||||
process.stderr.write("Syncing session files...\n");
|
||||
let lastWidth = 0;
|
||||
let lastRender = 0;
|
||||
const { processed, files } = await syncAllSessions({
|
||||
onProgress: event => {
|
||||
if (!tty) return;
|
||||
const now = Date.now();
|
||||
if (event.current < event.total && now - lastRender < 33) return;
|
||||
lastRender = now;
|
||||
const marker = "/sessions/";
|
||||
const idx = event.sessionFile.indexOf(marker);
|
||||
const short = idx >= 0 ? event.sessionFile.slice(idx + marker.length) : event.sessionFile;
|
||||
const pct = ((event.current / event.total) * 100).toFixed(0).padStart(3, " ");
|
||||
const line = `[${event.current}/${event.total}] ${pct}% ${short}`;
|
||||
const columns = process.stderr.columns ?? 120;
|
||||
const clipped = line.length > columns - 1 ? `${line.slice(0, columns - 2)}\u2026` : line;
|
||||
process.stderr.write(`\r${clipped.padEnd(lastWidth)}`);
|
||||
lastWidth = clipped.length;
|
||||
},
|
||||
});
|
||||
if (tty && lastWidth > 0) process.stderr.write(`\r${" ".repeat(lastWidth)}\r`);
|
||||
const total = await getTotalMessageCount();
|
||||
console.log(`Synced ${processed} new entries from ${files} files (${total} total)\n`);
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@ import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
||||
import { getSessionsDir, isEnoent } from "@oh-my-pi/pi-utils";
|
||||
import type { MessageStats, SessionEntry, SessionMessageEntry, UserMessageStats } from "./types";
|
||||
import type { MessageStats, SessionEntry, SessionMessageEntry, UserMessageLink, UserMessageStats } from "./types";
|
||||
import { computeUserMessageMetrics } from "./user-metrics";
|
||||
|
||||
/**
|
||||
@@ -71,9 +71,12 @@ function extractUserStats(sessionFile: string, folder: string, entry: SessionMes
|
||||
provider: null,
|
||||
chars: metrics.chars,
|
||||
words: metrics.words,
|
||||
yellingSentences: metrics.yellingSentences,
|
||||
yelling: metrics.yelling,
|
||||
profanity: metrics.profanity,
|
||||
dramaRuns: metrics.dramaRuns,
|
||||
anguish: metrics.anguish,
|
||||
negation: metrics.negation,
|
||||
repetition: metrics.repetition,
|
||||
blame: metrics.blame,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -131,21 +134,26 @@ function parseSessionEntriesLenient(bytes: Uint8Array): { entries: SessionEntry[
|
||||
* Parse a session file and extract all assistant message stats.
|
||||
* Uses incremental reading with offset tracking.
|
||||
*/
|
||||
export async function parseSessionFile(
|
||||
sessionPath: string,
|
||||
fromOffset = 0,
|
||||
): Promise<{ stats: MessageStats[]; userStats: UserMessageStats[]; newOffset: number }> {
|
||||
export interface ParseSessionResult {
|
||||
stats: MessageStats[];
|
||||
userStats: UserMessageStats[];
|
||||
userLinks: UserMessageLink[];
|
||||
newOffset: number;
|
||||
}
|
||||
|
||||
export async function parseSessionFile(sessionPath: string, fromOffset = 0): Promise<ParseSessionResult> {
|
||||
let bytes: Uint8Array;
|
||||
try {
|
||||
bytes = await Bun.file(sessionPath).bytes();
|
||||
} catch (err) {
|
||||
if (isEnoent(err)) return { stats: [], userStats: [], newOffset: fromOffset };
|
||||
if (isEnoent(err)) return { stats: [], userStats: [], userLinks: [], newOffset: fromOffset };
|
||||
throw err;
|
||||
}
|
||||
|
||||
const folder = extractFolderFromPath(sessionPath);
|
||||
const stats: MessageStats[] = [];
|
||||
const userStats: UserMessageStats[] = [];
|
||||
const userLinks: UserMessageLink[] = [];
|
||||
const userByEntryId = new Map<string, UserMessageStats>();
|
||||
const start = Math.max(0, Math.min(fromOffset, bytes.length));
|
||||
const unprocessed = bytes.subarray(start);
|
||||
@@ -165,17 +173,26 @@ export async function parseSessionFile(
|
||||
// Link assistant's responding model back to the user message it answered.
|
||||
const parentId = (entry as SessionMessageEntry).parentId;
|
||||
if (parentId) {
|
||||
const parentUser = userByEntryId.get(parentId);
|
||||
if (parentUser && parentUser.model === null) {
|
||||
const msg = entry.message as AssistantMessage;
|
||||
parentUser.model = msg.model;
|
||||
parentUser.provider = msg.provider;
|
||||
const msg = entry.message as AssistantMessage;
|
||||
if (msg.model && msg.provider) {
|
||||
// Emit unconditionally. The aggregator's UPDATE is guarded by
|
||||
// `model IS NULL` so this is idempotent: a no-op for already
|
||||
// linked rows, a fix-up for fresh inserts (which start NULL
|
||||
// because the user row is recorded before its reply lands) and
|
||||
// for cross-pass orphans whose parent was committed by an
|
||||
// earlier incremental sync.
|
||||
userLinks.push({
|
||||
sessionFile: sessionPath,
|
||||
entryId: parentId,
|
||||
model: msg.model,
|
||||
provider: msg.provider,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return { stats, userStats, newOffset: start + read };
|
||||
return { stats, userStats, userLinks, newOffset: start + read };
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
/**
|
||||
* Stateless parse worker for `syncAllSessions`. The main thread owns the
|
||||
* SQLite handle; workers receive `{ sessionFile, fromOffset }`, run
|
||||
* `parseSessionFile` (which is pure I/O + CPU, no DB), and post the
|
||||
* structured-clone-safe result back. One in-flight request per worker so
|
||||
* the main thread can fan jobs out 1:1 with the pool size.
|
||||
*/
|
||||
|
||||
import { type ParseSessionResult, parseSessionFile } from "./parser";
|
||||
|
||||
export interface SyncWorkerRequest {
|
||||
sessionFile: string;
|
||||
fromOffset: number;
|
||||
}
|
||||
|
||||
export type SyncWorkerResponse = { ok: true; result: ParseSessionResult } | { ok: false; error: string };
|
||||
|
||||
declare const self: Worker & {
|
||||
onmessage: ((event: MessageEvent<SyncWorkerRequest>) => void) | null;
|
||||
};
|
||||
|
||||
self.onmessage = async event => {
|
||||
const { sessionFile, fromOffset } = event.data;
|
||||
try {
|
||||
const result = await parseSessionFile(sessionFile, fromOffset);
|
||||
self.postMessage({ ok: true, result } satisfies SyncWorkerResponse);
|
||||
} catch (err) {
|
||||
const error = err instanceof Error ? (err.stack ?? err.message) : String(err);
|
||||
self.postMessage({ ok: false, error } satisfies SyncWorkerResponse);
|
||||
}
|
||||
};
|
||||
+42
-10
@@ -219,11 +219,31 @@ export interface UserMessageStats {
|
||||
/** Whitespace-delimited word count */
|
||||
words: number;
|
||||
/** Yelling sentences (> 50% uppercase letters) */
|
||||
yellingSentences: number;
|
||||
yelling: number;
|
||||
/** Profanity hits */
|
||||
profanity: number;
|
||||
/** Runs of 3+ consecutive `!` / `?` */
|
||||
dramaRuns: number;
|
||||
/** Catch-all upset signal: drama runs + `noooo`/`ughh`/... + `dude` + `..` */
|
||||
anguish: number;
|
||||
/** Corrective negation ("no", "nope", "thats not what i meant") */
|
||||
negation: number;
|
||||
/** User repeating themselves ("i meant", "still doesnt work", "like i said") */
|
||||
repetition: number;
|
||||
/** Second-person reproach ("you didnt", "you broke", "stop X-ing") */
|
||||
blame: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pair emitted by the parser when it sees an assistant message whose
|
||||
* `parentId` points to a user message that wasn't parsed in the same pass
|
||||
* (e.g. user prompt landed in an earlier incremental sync). The aggregator
|
||||
* applies the link to the persisted `user_messages` row so it stops showing
|
||||
* up in the "unknown" model bucket.
|
||||
*/
|
||||
export interface UserMessageLink {
|
||||
sessionFile: string;
|
||||
entryId: string;
|
||||
model: string;
|
||||
provider: string;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -239,20 +259,29 @@ export interface BehaviorTimeSeriesPoint {
|
||||
/** Number of user messages in bucket */
|
||||
messages: number;
|
||||
/** Total yelling sentences in bucket */
|
||||
yellingSentences: number;
|
||||
yelling: number;
|
||||
/** Total profanity hits in bucket */
|
||||
profanity: number;
|
||||
/** Total drama runs in bucket */
|
||||
dramaRuns: number;
|
||||
/** Total anguish signal in bucket */
|
||||
anguish: number;
|
||||
/** Total corrective-negation hits in bucket */
|
||||
negation: number;
|
||||
/** Total user-repeating-themselves hits in bucket */
|
||||
repetition: number;
|
||||
/** Total second-person blame hits in bucket */
|
||||
blame: number;
|
||||
/** Total characters in bucket */
|
||||
chars: number;
|
||||
}
|
||||
|
||||
export interface BehaviorOverallStats {
|
||||
totalMessages: number;
|
||||
totalYellingSentences: number;
|
||||
totalYelling: number;
|
||||
totalProfanity: number;
|
||||
totalDramaRuns: number;
|
||||
totalAnguish: number;
|
||||
totalNegation: number;
|
||||
totalRepetition: number;
|
||||
totalBlame: number;
|
||||
totalChars: number;
|
||||
firstTimestamp: number;
|
||||
lastTimestamp: number;
|
||||
@@ -265,9 +294,12 @@ export interface BehaviorModelStats {
|
||||
model: string;
|
||||
provider: string;
|
||||
totalMessages: number;
|
||||
totalYellingSentences: number;
|
||||
totalYelling: number;
|
||||
totalProfanity: number;
|
||||
totalDramaRuns: number;
|
||||
totalAnguish: number;
|
||||
totalNegation: number;
|
||||
totalRepetition: number;
|
||||
totalBlame: number;
|
||||
totalChars: number;
|
||||
lastTimestamp: number;
|
||||
}
|
||||
|
||||
@@ -13,13 +13,55 @@ export interface UserMessageMetrics {
|
||||
/**
|
||||
* Number of "yelling" sentences: sentences where more than half of the
|
||||
* alphabetic characters are uppercase (and there are enough letters to
|
||||
* make the ratio meaningful — short acronyms like "OK" don't count).
|
||||
* make the ratio meaningful - short acronyms like "OK" don't count).
|
||||
*/
|
||||
yellingSentences: number;
|
||||
yelling: number;
|
||||
/** Profanity hits (word-boundary, case-insensitive). */
|
||||
profanity: number;
|
||||
/** Runs of 3+ `!` / `?` characters (including `1`-mishit fallout). */
|
||||
dramaRuns: number;
|
||||
/**
|
||||
* Catch-all "obviously upset" signal computed on a *prose-only* body
|
||||
* (code fences, XML/HTML tags, URLs, file mentions, and quoted lines
|
||||
* are stripped first; messages whose remaining prose is >=3 lines score
|
||||
* zero because formatted prompts aren't tantrums).
|
||||
*
|
||||
* Sum of:
|
||||
* - drama runs: 3+ `!` / `?` (with `1`-mishit fallout)
|
||||
* - elongated interjections: `noooo`, `ahhhh`, `ughhh`, `argh`, `stooop`,
|
||||
* `whyyy`, `fuuu(ck)`, `shiiit`, `wtfff`, `omggg`, `yessss`, `helpp`,
|
||||
* `goddd`, `dammm`, `bruhh`
|
||||
* - standalone `dude`
|
||||
* - dot runs: `..`, `...`, `....+`
|
||||
*/
|
||||
anguish: number;
|
||||
/**
|
||||
* Corrective negation: the user is telling us we got it wrong.
|
||||
*
|
||||
* Counted on the same prose-only body as {@link anguish}.
|
||||
*
|
||||
* - line-leading `no` / `nope` / `nah` / `nvm` / `wrong` / `incorrect`
|
||||
* (word-bounded, so `now`, `nobody`, `north` don't match)
|
||||
* - `that(?:'s)? not (what|right|it)` and `not what i (meant|asked|said|wanted)`
|
||||
*/
|
||||
negation: number;
|
||||
/**
|
||||
* The user is repeating themselves - strong signal the previous turn
|
||||
* missed the ask. Counts hits for:
|
||||
*
|
||||
* - `i (meant|said|asked|told you|already (said|told|did|asked|wrote))`
|
||||
* - `(like|as) i (said|told you|asked)`
|
||||
* - `still (doesn't|isn't|not|broken|wrong|fails|failing|the same|same)`
|
||||
*
|
||||
* Bare `still` / `again` are too ambiguous to count alone (they show up
|
||||
* in normal speech like "try again" or "still works").
|
||||
*/
|
||||
repetition: number;
|
||||
/**
|
||||
* Direct second-person reproach pinned on the agent:
|
||||
*
|
||||
* - `you (didn't|did not|broke|missed|forgot|keep|always|never|still|ignored)`
|
||||
* - sentence-leading `stop <verb>ing` imperatives
|
||||
*/
|
||||
blame: number;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -363,15 +405,20 @@ const PROFANITY: readonly string[] = [
|
||||
"garbage",
|
||||
"crud",
|
||||
"crudded",
|
||||
// quality-dismissal ("this is garbage / pointless")
|
||||
"useless",
|
||||
"pointless",
|
||||
"horrible",
|
||||
"awful",
|
||||
"worthless",
|
||||
"ridiculous",
|
||||
"nonsense",
|
||||
// religious exclamations
|
||||
"jesus",
|
||||
"christ",
|
||||
"jeez",
|
||||
"jeezus",
|
||||
"sheesh",
|
||||
"holymoly",
|
||||
"holyfuck",
|
||||
"holysmokes",
|
||||
"godsake",
|
||||
// chat acronyms
|
||||
"wtf",
|
||||
@@ -415,18 +462,98 @@ const PROFANITY: readonly string[] = [
|
||||
"grrrr",
|
||||
];
|
||||
|
||||
const PROFANITY_RE = new RegExp(`\\b(?:${PROFANITY.join("|")})\\b`, "gi");
|
||||
const PROFANITY_RE = new RegExp(String.raw`\b(?:${PROFANITY.join("|")})\b`, "gi");
|
||||
const SENTENCE_RE = /[^.!?\n]+/g;
|
||||
const LETTER_RE = /\p{L}/gu;
|
||||
const UPPER_LETTER_RE = /\p{Lu}/gu;
|
||||
const YELLING_MIN_LETTERS = 4;
|
||||
const YELLING_THRESHOLD = 0.5;
|
||||
// Runs starting with `!` or `?` followed by ≥2 of `!?1`. The `1` is the
|
||||
// Runs starting with `!` or `?` followed by 2+ of `!?1`. The `1` is the
|
||||
// classic shift-key mishit ("!!!111" / "!?!??111") so we count those as
|
||||
// part of the same drama burst.
|
||||
const DRAMA_RE = /[!?][!?1]{2,}/g;
|
||||
const WORD_RE = /\S+/g;
|
||||
|
||||
// Elongated anguish/exasperation interjections. Each alternative is a
|
||||
// case-insensitive word-bounded pattern that requires *real* elongation
|
||||
// (so plain "no" / "argh" / "ahh" / "god" don't fire). Picked to avoid
|
||||
// hex / base64 contamination via the surrounding `\b` plus letter-only
|
||||
// alternatives.
|
||||
const ANGUISH_PATTERNS: readonly string[] = [
|
||||
"no{3,}", // nooo, noooooo
|
||||
"a+h{2,}", // ahh, aaaahhh
|
||||
"u+g+h{2,}", // ughh, uuugh
|
||||
"a+r+g+h+", // argh, aaargh, arrgghhh
|
||||
"st+o{3,}p+", // stooop, sttooopp
|
||||
"w+h+y{3,}", // whyyy, whyyyyy
|
||||
"f+u{3,}c*k*", // fuuu, fuuuck
|
||||
"wtf{3,}", // wtfff
|
||||
"o+m+g{2,}", // omgg, omggg
|
||||
"ye+s{3,}", // yesss, yeessss
|
||||
"g+o+d{3,}", // goddd, goddddd
|
||||
"br+u+h{2,}", // bruhh, bruuuhh
|
||||
];
|
||||
const ANGUISH_RE = new RegExp(String.raw`\b(?:${ANGUISH_PATTERNS.join("|")})\b`, "gi");
|
||||
const DUDE_RE = /\bdude\b/gi;
|
||||
// Runs of 2+ dots. Captures `..` (lazy trail-off), `...` (tentative
|
||||
// ellipsis), and `....+` (exasperation) in a single signal.
|
||||
const ELLIPSIS_RE = /\.{2,}/g;
|
||||
|
||||
// --- Frustration signals ----------------------------------------------------
|
||||
// Each set of patterns below is tuned against ~42k real user prompts so the
|
||||
// short-prose hits are dominated by genuine frustration, not technical talk.
|
||||
|
||||
// Corrective negation. We deliberately anchor to the very start of the
|
||||
// trimmed prose body (no `m` flag) - in practice mid-message lines that
|
||||
// start with `no`/`Wrong`/`No JSDoc warning` are list items, pasted error
|
||||
// text or descriptive statements, not actual corrections. Real frustration
|
||||
// negation overwhelmingly opens the message.
|
||||
const NEGATION_LEAD_RE = /^[ \t]*(?:no|nope|nah|nvm|wrong|incorrect)\b/gi;
|
||||
const NEGATION_PHRASE_RE =
|
||||
/\b(?:that['\u2019]?s\s+not\s+(?:what|right|it)|not\s+what\s+i\s+(?:meant|asked|said|wanted))\b/gi;
|
||||
|
||||
// User repeating themselves. The recall pattern accepts an optional
|
||||
// `like ` / `as ` prefix so "like i said" doesn't double-count with bare
|
||||
// "i said". Bare `i asked` is too noisy - it's overwhelmingly "i asked
|
||||
// <some third party>" in this corpus (committee, experts, weaker LLM, ...) -
|
||||
// so we require `i asked you` for that variant. Bare `still` / `again` are
|
||||
// ambiguous so we only count `still` when followed by a negative or
|
||||
// sameness marker.
|
||||
const REPETITION_RECALL_RE =
|
||||
/\b(?:(?:like|as)\s+i\s+(?:said|told\s+you|asked)|i\s+(?:meant|said|told\s+you|asked\s+you|already\s+(?:said|told|did|asked|wrote)))\b/gi;
|
||||
const REPETITION_STILL_RE =
|
||||
/\bstill\s+(?:doesn['\u2019]?t|doesnt|isn['\u2019]?t|isnt|not|broken|wrong|fails|failing|the\s+same|same)\b/gi;
|
||||
|
||||
// Direct second-person reproach. `you` alone is too generic (>7k hits in
|
||||
// short prose), so we anchor it to a small set of accusatory verbs.
|
||||
const BLAME_YOU_RE = /\byou\s+(?:didn['\u2019]?t|did\s+not|broke|missed|forgot|keep|always|never|still|ignored)\b/gi;
|
||||
// `stop <verb>ing` is only frustration when it's an imperative - require it
|
||||
// to start a sentence (line start or after a sentence-terminating punctuator).
|
||||
const BLAME_STOP_RE = /(?:^|(?<=[.!?\n]))\s*stop\s+\w+ing\b/gim;
|
||||
|
||||
// Stripped from the analyzed body before scoring so that structured
|
||||
// content (code, XML/HTML, URLs, file mentions, quoted blocks) doesn't
|
||||
// pollute behavior signals. We replace with a newline so line counts
|
||||
// reflect what was removed instead of merging neighbors.
|
||||
const FENCED_CODE_RE = /```[\s\S]*?```/g;
|
||||
const XML_TAG_PAIR_RE = /<([A-Za-z][\w-]*)\b[^>]*>[\s\S]*?<\/\1>/g;
|
||||
const XML_TAG_BARE_RE = /<\/?[A-Za-z][\w-]*\b[^>]*\/?>/g;
|
||||
const INLINE_CODE_RE = /`[^`\n]*`/g;
|
||||
const URL_RE = /\bhttps?:\/\/\S+/gi;
|
||||
const FILE_MENTION_RE = /(^|\s)@[\w./-]+/g;
|
||||
const QUOTE_LINE_RE = /^[ \t]*>.*$/gm;
|
||||
// Harness placeholders the TUI substitutes for binary/non-text user input.
|
||||
// Strip them so real frustration signals on later lines aren't masked off
|
||||
// by `[Image #1]` etc. consuming line 1.
|
||||
const IMAGE_MARKER_RE = /\[Image #\d+\]/g;
|
||||
// ANSI escape sequences sometimes leak in from terminal copy-paste
|
||||
// (e.g. when the user pastes a bash transcript). Strip them.
|
||||
const ANSI_ESCAPE_RE = /\x1b\[[0-9;]*[A-Za-z]/g;
|
||||
|
||||
// Users don't really get angry with super detailed and formatted prompts
|
||||
// - if the remaining prose is this many lines or more, score zero.
|
||||
const MAX_PROSE_LINES = 3;
|
||||
|
||||
/** Count regex hits without materializing the match array. */
|
||||
function countMatches(text: string, re: RegExp): number {
|
||||
let count = 0;
|
||||
@@ -457,6 +584,33 @@ function countYellingSentences(text: string): number {
|
||||
return count;
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip structured content so that pasted code, harness wrappers, file
|
||||
* mentions and quoted blocks don't dilute or fake behavior signals.
|
||||
* Each strip is replaced with a newline so subsequent line counting
|
||||
* reflects what was removed instead of merging neighbors.
|
||||
*/
|
||||
function stripStructuredContent(text: string): string {
|
||||
return text
|
||||
.replace(FENCED_CODE_RE, "\n")
|
||||
.replace(XML_TAG_PAIR_RE, "\n")
|
||||
.replace(XML_TAG_BARE_RE, " ")
|
||||
.replace(INLINE_CODE_RE, " ")
|
||||
.replace(URL_RE, " ")
|
||||
.replace(FILE_MENTION_RE, "$1 ")
|
||||
.replace(QUOTE_LINE_RE, "")
|
||||
.replace(IMAGE_MARKER_RE, " ")
|
||||
.replace(ANSI_ESCAPE_RE, "");
|
||||
}
|
||||
|
||||
function countNonEmptyLines(text: string): number {
|
||||
let count = 0;
|
||||
for (const line of text.split("\n")) {
|
||||
if (line.trim().length > 0) count++;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute behavioral metrics for a user message.
|
||||
*
|
||||
@@ -465,14 +619,57 @@ function countYellingSentences(text: string): number {
|
||||
export function computeUserMessageMetrics(text: string): UserMessageMetrics {
|
||||
const trimmed = text.trim();
|
||||
if (!trimmed) {
|
||||
return { chars: 0, words: 0, yellingSentences: 0, profanity: 0, dramaRuns: 0 };
|
||||
return {
|
||||
chars: 0,
|
||||
words: 0,
|
||||
yelling: 0,
|
||||
profanity: 0,
|
||||
anguish: 0,
|
||||
negation: 0,
|
||||
repetition: 0,
|
||||
blame: 0,
|
||||
};
|
||||
}
|
||||
|
||||
const chars = trimmed.length;
|
||||
const words = countMatches(trimmed, WORD_RE);
|
||||
|
||||
// Behavior signals are computed on a stripped prose body; long /
|
||||
// well-formatted messages score zero because they are deliberate, not
|
||||
// emotional outbursts.
|
||||
const prose = stripStructuredContent(trimmed).trim();
|
||||
if (!prose || countNonEmptyLines(prose) >= MAX_PROSE_LINES) {
|
||||
return {
|
||||
chars,
|
||||
words,
|
||||
yelling: 0,
|
||||
profanity: 0,
|
||||
anguish: 0,
|
||||
negation: 0,
|
||||
repetition: 0,
|
||||
blame: 0,
|
||||
};
|
||||
}
|
||||
|
||||
const anguish =
|
||||
countMatches(prose, DRAMA_RE) +
|
||||
countMatches(prose, ANGUISH_RE) +
|
||||
countMatches(prose, DUDE_RE) +
|
||||
countMatches(prose, ELLIPSIS_RE);
|
||||
|
||||
const negation = countMatches(prose, NEGATION_LEAD_RE) + countMatches(prose, NEGATION_PHRASE_RE);
|
||||
const repetition = countMatches(prose, REPETITION_RECALL_RE) + countMatches(prose, REPETITION_STILL_RE);
|
||||
const blame = countMatches(prose, BLAME_YOU_RE) + countMatches(prose, BLAME_STOP_RE);
|
||||
|
||||
return {
|
||||
chars: trimmed.length,
|
||||
words: countMatches(trimmed, WORD_RE),
|
||||
yellingSentences: countYellingSentences(trimmed),
|
||||
profanity: countMatches(trimmed, PROFANITY_RE),
|
||||
dramaRuns: countMatches(trimmed, DRAMA_RE),
|
||||
chars,
|
||||
words,
|
||||
yelling: countYellingSentences(prose),
|
||||
profanity: countMatches(prose, PROFANITY_RE),
|
||||
anguish,
|
||||
negation,
|
||||
repetition,
|
||||
blame,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -480,7 +677,10 @@ export function computeUserMessageMetrics(text: string): UserMessageMetrics {
|
||||
export const EMPTY_USER_METRICS: UserMessageMetrics = Object.freeze({
|
||||
chars: 0,
|
||||
words: 0,
|
||||
yellingSentences: 0,
|
||||
yelling: 0,
|
||||
profanity: 0,
|
||||
dramaRuns: 0,
|
||||
anguish: 0,
|
||||
negation: 0,
|
||||
repetition: 0,
|
||||
blame: 0,
|
||||
});
|
||||
|
||||
@@ -1,85 +1,176 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { computeUserMessageMetrics } from "../src/user-metrics";
|
||||
import { computeUserMessageMetrics, EMPTY_USER_METRICS } from "../src/user-metrics";
|
||||
|
||||
describe("computeUserMessageMetrics", () => {
|
||||
it("returns zeros for empty / whitespace-only text", () => {
|
||||
expect(computeUserMessageMetrics("")).toEqual({
|
||||
chars: 0,
|
||||
words: 0,
|
||||
yellingSentences: 0,
|
||||
profanity: 0,
|
||||
dramaRuns: 0,
|
||||
});
|
||||
expect(computeUserMessageMetrics(" \n\t ")).toEqual({
|
||||
chars: 0,
|
||||
words: 0,
|
||||
yellingSentences: 0,
|
||||
profanity: 0,
|
||||
dramaRuns: 0,
|
||||
});
|
||||
expect(computeUserMessageMetrics("")).toEqual({ ...EMPTY_USER_METRICS });
|
||||
expect(computeUserMessageMetrics(" \n\t ")).toEqual({ ...EMPTY_USER_METRICS });
|
||||
});
|
||||
|
||||
it("counts a sentence as yelling when >50% of its letters are uppercase", () => {
|
||||
// 16 letters, all uppercase → yelling.
|
||||
const m = computeUserMessageMetrics("STOP DOING THAT NOW");
|
||||
expect(m.yellingSentences).toBe(1);
|
||||
expect(m.yelling).toBe(1);
|
||||
});
|
||||
|
||||
it("treats mostly-lowercase sentences as not yelling even with embedded CAPS", () => {
|
||||
// `STOP` and `THAT` are uppercase but the surrounding lowercase keeps the
|
||||
// per-sentence ratio well under 50%.
|
||||
const m = computeUserMessageMetrics("Hi there, please STOP doing THAT immediately, it is really annoying.");
|
||||
expect(m.yellingSentences).toBe(0);
|
||||
expect(m.yelling).toBe(0);
|
||||
});
|
||||
|
||||
it("ignores very short uppercase fragments below the letter floor", () => {
|
||||
// "OK" and "WIP" have fewer than the minimum-letters threshold, so neither
|
||||
// sentence should register as yelling.
|
||||
expect(computeUserMessageMetrics("OK").yellingSentences).toBe(0);
|
||||
expect(computeUserMessageMetrics("WIP.").yellingSentences).toBe(0);
|
||||
expect(computeUserMessageMetrics("OK").yelling).toBe(0);
|
||||
expect(computeUserMessageMetrics("WIP.").yelling).toBe(0);
|
||||
});
|
||||
|
||||
it("counts multiple yelling sentences separated by terminators", () => {
|
||||
const m = computeUserMessageMetrics("WHY IS THIS BROKEN? FIX IT NOW!! please.");
|
||||
// "WHY IS THIS BROKEN" and " FIX IT NOW" are both >50% uppercase; the
|
||||
// trailing " please" sentence is lowercase.
|
||||
expect(m.yellingSentences).toBe(2);
|
||||
expect(m.yelling).toBe(2);
|
||||
});
|
||||
|
||||
it("does not flag camelCase / acronyms inside otherwise-lowercase prose", () => {
|
||||
const m = computeUserMessageMetrics("call getHTMLParser then exit");
|
||||
expect(m.yellingSentences).toBe(0);
|
||||
expect(m.yelling).toBe(0);
|
||||
});
|
||||
|
||||
it("matches profanity case-insensitively at word boundaries only", () => {
|
||||
// Regression: prior version used a non-raw template literal so `\b` was
|
||||
// compiled as backspace (U+0008) and the regex matched nothing in real
|
||||
// prose. Lock the word-boundary contract in.
|
||||
const m = computeUserMessageMetrics("oh FUCK this is bullshit, damn it");
|
||||
expect(m.profanity).toBe(3);
|
||||
// `class` shares letters with `ass` but must not match — word boundary required.
|
||||
expect(computeUserMessageMetrics("import classes from module").profanity).toBe(0);
|
||||
});
|
||||
|
||||
it("counts each run of 3+ ! or ? as one drama run", () => {
|
||||
it("counts quality-dismissal vocabulary as profanity", () => {
|
||||
const m = computeUserMessageMetrics("this is garbage, useless and horrible work");
|
||||
expect(m.profanity).toBe(3);
|
||||
});
|
||||
|
||||
it("folds drama runs / elongated interjections / dot trails into `anguish`", () => {
|
||||
const m = computeUserMessageMetrics("why!!! seriously??? omg!?!?!?");
|
||||
// "!!!" = 1, "???" = 1, "!?!?!?" = 1 (mixed ≥3 cluster) → 3
|
||||
expect(m.dramaRuns).toBe(3);
|
||||
// Two characters alone do not count as drama.
|
||||
expect(computeUserMessageMetrics("ok!! sure??").dramaRuns).toBe(0);
|
||||
expect(m.anguish).toBeGreaterThanOrEqual(3);
|
||||
expect(computeUserMessageMetrics("ok!! sure??").anguish).toBe(0);
|
||||
});
|
||||
|
||||
it("absorbs shift-key `1` mishits into the surrounding drama run", () => {
|
||||
// "!!!111" and "!?!?!??111" are both single bursts, not separate hits.
|
||||
expect(computeUserMessageMetrics("what!!!111").dramaRuns).toBe(1);
|
||||
expect(computeUserMessageMetrics("are you serious!?!?!??111").dramaRuns).toBe(1);
|
||||
// Plain digits without a leading `!`/`?` are not drama.
|
||||
expect(computeUserMessageMetrics("port 8111 please").dramaRuns).toBe(0);
|
||||
it("absorbs shift-key `1` mishits into a single drama burst", () => {
|
||||
expect(computeUserMessageMetrics("what!!!111").anguish).toBeGreaterThanOrEqual(1);
|
||||
expect(computeUserMessageMetrics("are you serious!?!?!??111").anguish).toBeGreaterThanOrEqual(1);
|
||||
expect(computeUserMessageMetrics("port 8111 please").anguish).toBe(0);
|
||||
});
|
||||
|
||||
it("captures all three signals together with correct chars/words", () => {
|
||||
const m = computeUserMessageMetrics("WHY IS THIS SO SHITTY???");
|
||||
expect(m.yellingSentences).toBe(1);
|
||||
expect(m.profanity).toBe(1);
|
||||
expect(m.dramaRuns).toBe(1);
|
||||
expect(m.words).toBe(5);
|
||||
expect(m.chars).toBe("WHY IS THIS SO SHITTY???".length);
|
||||
describe("negation signal", () => {
|
||||
it("fires on line-leading correction openers", () => {
|
||||
expect(computeUserMessageMetrics("no this is the renderer").negation).toBe(1);
|
||||
expect(computeUserMessageMetrics("nope, still wrong").negation).toBe(1);
|
||||
expect(computeUserMessageMetrics("nah look at this").negation).toBe(1);
|
||||
expect(computeUserMessageMetrics("wrong file").negation).toBe(1);
|
||||
expect(computeUserMessageMetrics("nvm got it").negation).toBe(1);
|
||||
});
|
||||
|
||||
it("does not fire on words that share a prefix with negation tokens", () => {
|
||||
// `now`, `nobody`, `north`, `noble`, `normal` all start with `no` but
|
||||
// are not corrective negation.
|
||||
expect(computeUserMessageMetrics("now everything works").negation).toBe(0);
|
||||
expect(computeUserMessageMetrics("nobody knows why").negation).toBe(0);
|
||||
expect(computeUserMessageMetrics("normal operation resumed").negation).toBe(0);
|
||||
});
|
||||
|
||||
it("only anchors at the start of the message - mid-message `no`/`No` lines do not fire", () => {
|
||||
// Real corrective negation overwhelmingly opens the message. Pasted error
|
||||
// text and bullet lists trip the old `^...$/m` anchor with FPs like
|
||||
// "Wrong user name or password" or "No JSDoc warning on X".
|
||||
expect(computeUserMessageMetrics("i instantly get Finalizing ->\nNo speech detected").negation).toBe(0);
|
||||
expect(computeUserMessageMetrics("Authentication failed\n\nWrong user name or password").negation).toBe(0);
|
||||
});
|
||||
|
||||
it("strips `[Image #N]` placeholders so message-leading negation still fires", () => {
|
||||
// The TUI inserts `[Image #1]` markers ahead of real user prose; the
|
||||
// strip pass removes them so anchored-at-start negation still works.
|
||||
expect(computeUserMessageMetrics("[Image #1] nope still broken").negation).toBe(1);
|
||||
});
|
||||
|
||||
it("fires on explicit rejection phrases", () => {
|
||||
expect(computeUserMessageMetrics("thats not what i wanted").negation).toBe(1);
|
||||
expect(computeUserMessageMetrics("that's not right").negation).toBe(1);
|
||||
expect(computeUserMessageMetrics("this is not what i meant at all").negation).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("repetition signal", () => {
|
||||
it("counts explicit recall verbs", () => {
|
||||
expect(computeUserMessageMetrics("i meant the other file").repetition).toBe(1);
|
||||
expect(computeUserMessageMetrics("i told you to skip it").repetition).toBe(1);
|
||||
expect(computeUserMessageMetrics("i asked you for json not yaml").repetition).toBe(1);
|
||||
expect(computeUserMessageMetrics("like i said earlier").repetition).toBe(1);
|
||||
});
|
||||
|
||||
it("requires `you` after `i asked` to suppress neutral third-party usage", () => {
|
||||
// In the real corpus, bare `i asked` is overwhelmingly "i asked
|
||||
// <some third party>" - committee, experts, weaker LLMs, etc. -
|
||||
// which is not frustration with us. The `(like|as) i asked` form is
|
||||
// still allowed because it always refers back to our own ask.
|
||||
expect(computeUserMessageMetrics("i asked the committee to review").repetition).toBe(0);
|
||||
expect(computeUserMessageMetrics("so i asked a bunch of experts").repetition).toBe(0);
|
||||
expect(computeUserMessageMetrics("you're not doing AST rewriting like i asked").repetition).toBe(1);
|
||||
});
|
||||
|
||||
it("counts `still` only when paired with a negative / sameness marker", () => {
|
||||
// Bare `still` would over-fire on neutral usage.
|
||||
expect(computeUserMessageMetrics("the agent still works fine").repetition).toBe(0);
|
||||
expect(computeUserMessageMetrics("it still doesnt work").repetition).toBe(1);
|
||||
expect(computeUserMessageMetrics("still the same issue").repetition).toBe(1);
|
||||
expect(computeUserMessageMetrics("still failing on darwin").repetition).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("blame signal", () => {
|
||||
it("fires on accusatory second-person verbs", () => {
|
||||
expect(computeUserMessageMetrics("you broke the layout").blame).toBe(1);
|
||||
expect(computeUserMessageMetrics("you didnt update AGENTS").blame).toBe(1);
|
||||
expect(computeUserMessageMetrics("you missed a callsite").blame).toBe(1);
|
||||
expect(computeUserMessageMetrics("you forgot to commit").blame).toBe(1);
|
||||
expect(computeUserMessageMetrics("you keep doing that").blame).toBe(1);
|
||||
});
|
||||
|
||||
it("does not fire on bare `you`", () => {
|
||||
// `you` alone is too generic - dominated by neutral instructions.
|
||||
expect(computeUserMessageMetrics("can you fix the bug?").blame).toBe(0);
|
||||
expect(computeUserMessageMetrics("could you also add a test").blame).toBe(0);
|
||||
});
|
||||
|
||||
it("only fires on `stop X-ing` at sentence start", () => {
|
||||
expect(computeUserMessageMetrics("stop touching git").blame).toBe(1);
|
||||
expect(computeUserMessageMetrics("please stop making yolo changes").blame).toBe(0); // mid-sentence
|
||||
expect(computeUserMessageMetrics("ok. stop reverting things").blame).toBe(1);
|
||||
// `nonstop`/`stopping` should not match the imperative pattern.
|
||||
expect(computeUserMessageMetrics("the loop keeps stopping").blame).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
it("zeros out behavior signals on long structured prompts", () => {
|
||||
// >= 3 non-empty prose lines after stripping = deliberate prompt, not a tantrum.
|
||||
const long = [
|
||||
"no this is wrong, you broke it, i meant the other one.",
|
||||
"please undo and try again.",
|
||||
"acceptance: green tests.",
|
||||
"thanks!",
|
||||
].join("\n");
|
||||
const m = computeUserMessageMetrics(long);
|
||||
expect(m.negation).toBe(0);
|
||||
expect(m.repetition).toBe(0);
|
||||
expect(m.blame).toBe(0);
|
||||
expect(m.anguish).toBe(0);
|
||||
expect(m.profanity).toBe(0);
|
||||
expect(m.yelling).toBe(0);
|
||||
// But char/word counts still reflect the raw text.
|
||||
expect(m.chars).toBeGreaterThan(0);
|
||||
expect(m.words).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("captures multiple frustration signals on a single short message", () => {
|
||||
const m = computeUserMessageMetrics("no, you broke it AGAIN. i told you it still doesnt work");
|
||||
expect(m.negation).toBe(1);
|
||||
expect(m.blame).toBe(1);
|
||||
expect(m.repetition).toBeGreaterThanOrEqual(2); // `i told you` + `still doesnt`
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user