From b522fde56d2a8c4ed423cf08b0c22ea29f0c2eac Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 2 Jun 2026 05:24:21 +0200 Subject: [PATCH] perf(sqlite-reader): replaced full COUNT(*) scan with bounded row probing - Added ROW_COUNT_PROBE_CAP to limit rows scanned when counting tables, preventing JS thread freezes on large databases. - Used sqlite_stat1 estimates for tables exceeding the cap; exact counts only for provably small tables. - Introduced TableRowCount type with exact/estimate/atLeast variants reflected in rendered output. --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/tools/sqlite-reader.ts | 101 ++++++++++++++++-- .../coding-agent/test/tools/sqlite.test.ts | 86 ++++++++++++++- 3 files changed, 178 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 2d80adafc..f48eaad89 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed `read ` freezing the TUI on large databases. Listing tables ran an unbounded `SELECT COUNT(*)` per table, and since `bun:sqlite` executes synchronously on the same JS thread that drives rendering and input, a multi-GB database's full-table scans blocked the UI for seconds. The listing now reads the planner's `sqlite_stat1` estimate for tables above a scan cap (shown as `~N rows`) and only counts exactly when a table is provably small, reading at most `cap + 1` rows (a capped table shows `N+ rows`). On an 8.4 GB stats database the listing dropped from multi-second full scans to ~2 ms. - Fixed a module-load crash (`ReferenceError: Cannot access 'evalToolRenderer' before initialization`) triggered whenever `tools/eval` was imported before `tools/renderers`. The eval JS backend statically pulls the agent/task/sdk/extension chain, which re-enters the root barrel → `modes/components` → `tool-execution` → `renderers` while `eval.ts` was still initializing, so `renderers.ts` read `evalToolRenderer` in its TDZ. The eval TUI renderer is now split into a dependency-light `tools/eval-render.ts` that `renderers.ts` imports directly (decoupling pure rendering from the eval runtime); `eval.ts` re-exports `evalToolRenderer`/`EVAL_DEFAULT_PREVIEW_LINES` for compatibility. - Fixed `history.db` never recording the originating session id: the `session_id` column documented for 15.6.0 was missing from the shipped storage layer, so the column was never created/populated on the write path and every prompt row had `session_id` `NULL`. Restored the `session_id` column, schema migration (`ALTER TABLE history ADD COLUMN session_id` for pre-existing databases), and `HistoryEntry.sessionId`; wired interactive mode to register `setSessionResolver(...)` so prompts are stamped with the session active at submission time (tracking fork/resume switches); and re-enabled prompt-history ranking in the `--resume` and in-session session pickers via `HistoryStorage.matchingSessionIds()`. diff --git a/packages/coding-agent/src/tools/sqlite-reader.ts b/packages/coding-agent/src/tools/sqlite-reader.ts index 451f2ae5f..37948a48d 100644 --- a/packages/coding-agent/src/tools/sqlite-reader.ts +++ b/packages/coding-agent/src/tools/sqlite-reader.ts @@ -12,6 +12,15 @@ const MAX_QUERY_LIMIT = 500; const MAX_RENDER_WIDTH = 120; const MAX_COLUMN_WIDTH = 40; const MIN_COLUMN_WIDTH = 1; +/** + * Upper bound on rows scanned when counting a table for the listing. SQLite has + * no stored row count, so `COUNT(*)` is a full b-tree scan — multi-second on a + * multi-GB database, and `bun:sqlite` runs it synchronously on the JS thread + * that also drives the TUI, freezing rendering and input. The listing instead + * trusts the planner's `sqlite_stat1` estimate for large tables and only counts + * exactly when a table is provably small, reading at most this many rows. + */ +const ROW_COUNT_PROBE_CAP = 50_000; type SqliteBinding = Exclude>; @@ -26,6 +35,11 @@ interface SqliteCountRow { count: number; } +interface SqliteStat1Row { + tbl: string; + stat: string | null; +} + interface SqliteTableInfoRow { cid: number; name: string; @@ -50,6 +64,23 @@ export type SqliteSelector = export type SqliteRowLookup = { kind: "pk"; column: string; type: string } | { kind: "rowid" }; +/** + * Row count for a table in the listing. + * - `exact`: counted in full (the table is small enough to count cheaply). + * - `estimate`: the planner's `sqlite_stat1` figure; the table is too large to + * scan, so this may be stale. + * - `atLeast`: a lower bound; counting was capped before reaching the end. + */ +export type TableRowCount = + | { kind: "exact"; rows: number } + | { kind: "estimate"; rows: number } + | { kind: "atLeast"; rows: number }; + +export interface SqliteTableSummary { + name: string; + count: TableRowCount; +} + function splitSqliteRemainder(remainder: string): { subPath: string; queryString: string } { const queryIndex = remainder.indexOf("?"); if (queryIndex === -1) { @@ -495,20 +526,61 @@ export function parseSqliteSelector(subPath: string, queryString: string): Sqlit return { kind: "schema", table, sampleLimit: DEFAULT_SCHEMA_SAMPLE_LIMIT }; } -export function listTables(db: Database): { name: string; rowCount: number }[] { +/** + * Reads the planner's per-table row estimate from `sqlite_stat1` (populated by + * `ANALYZE`). The first integer of each `stat` string is the number of rows in + * that index; for a full (non-partial) index it equals the table's row count, + * so the max across a table's entries is the table estimate. Returns an empty + * map when the database was never analyzed. One small indexed read — no scan. + */ +function loadRowEstimates(db: Database): Map { + const estimates = new Map(); + const hasStat1 = db + .prepare, []>( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'sqlite_stat1'", + ) + .get(); + if (!hasStat1) return estimates; + + for (const { tbl, stat } of db.prepare("SELECT tbl, stat FROM sqlite_stat1").all()) { + if (!stat) continue; + const rows = Number.parseInt(stat, 10); + if (!Number.isFinite(rows)) continue; + const prev = estimates.get(tbl); + if (prev === undefined || rows > prev) estimates.set(tbl, rows); + } + return estimates; +} + +/** + * Counts a table while reading at most `cap + 1` rows. Returns an exact count + * when the table holds `cap` rows or fewer, otherwise a lower bound of `cap`. + * Bounds the worst-case scan so a stale or missing estimate can never trigger a + * full-table scan on the JS thread. + */ +function probeRowCount(db: Database, table: string, cap: number): TableRowCount { + const sql = `SELECT COUNT(*) AS count FROM (SELECT 1 FROM ${quoteSqliteIdentifier(table)} LIMIT ${cap + 1})`; + const counted = db.prepare(sql).get()?.count ?? 0; + return counted > cap ? { kind: "atLeast", rows: cap } : { kind: "exact", rows: counted }; +} + +export function listTables(db: Database, options: { probeCap?: number } = {}): SqliteTableSummary[] { + const cap = options.probeCap ?? ROW_COUNT_PROBE_CAP; const names = db .prepare, []>( "SELECT name FROM sqlite_master WHERE type = 'table' AND name NOT LIKE 'sqlite_%' ORDER BY name COLLATE NOCASE", ) .all(); + const estimates = loadRowEstimates(db); return names.map(({ name }) => { - const countRow = - db.prepare(`SELECT COUNT(*) AS count FROM ${quoteSqliteIdentifier(name)}`).get() ?? null; - return { - name, - rowCount: countRow?.count ?? 0, - }; + const estimate = estimates.get(name); + // Trust the planner only when it says the table is too large to count + // cheaply; otherwise count exactly (bounded), which also corrects a + // stale-low estimate without ever scanning more than `cap` rows. + const count: TableRowCount = + estimate !== undefined && estimate > cap ? { kind: "estimate", rows: estimate } : probeRowCount(db, name, cap); + return { name, count }; }); } @@ -679,13 +751,24 @@ export function deleteRowByRowId(db: Database, table: string, key: string): numb return statement.run(binding).changes; } -export function renderTableList(tables: { name: string; rowCount: number }[]): string { +function formatRowCount(count: TableRowCount): string { + switch (count.kind) { + case "exact": + return `${count.rows} rows`; + case "estimate": + return `~${count.rows} rows`; + case "atLeast": + return `${count.rows}+ rows`; + } +} + +export function renderTableList(tables: SqliteTableSummary[]): string { if (tables.length === 0) { return "(no tables)"; } return tables - .map(table => truncateToWidth(replaceTabs(`${table.name} (${table.rowCount} rows)`), MAX_RENDER_WIDTH)) + .map(table => truncateToWidth(replaceTabs(`${table.name} (${formatRowCount(table.count)})`), MAX_RENDER_WIDTH)) .join("\n"); } diff --git a/packages/coding-agent/test/tools/sqlite.test.ts b/packages/coding-agent/test/tools/sqlite.test.ts index 01a4f3d1f..98e847ebb 100644 --- a/packages/coding-agent/test/tools/sqlite.test.ts +++ b/packages/coding-agent/test/tools/sqlite.test.ts @@ -6,7 +6,13 @@ import * as path from "node:path"; import "../../src/tools/renderers"; import { Settings } from "../../src/config/settings"; import { ReadTool } from "../../src/tools/read"; -import { parseSqlitePathCandidates, parseSqliteSelector, renderTable } from "../../src/tools/sqlite-reader"; +import { + listTables, + parseSqlitePathCandidates, + parseSqliteSelector, + renderTable, + renderTableList, +} from "../../src/tools/sqlite-reader"; import { WriteTool } from "../../src/tools/write"; type ToolTextResult = { @@ -418,3 +424,81 @@ describe("SQLite tool support", () => { ).rejects.toThrow(/no column named 'bogus'/i); }); }); + +describe("SQLite table listing row counts", () => { + let tmpDir: string; + let dbPath: string; + + beforeEach(async () => { + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "sqlite-count-test-")); + dbPath = path.join(tmpDir, "counts.db"); + }); + + afterEach(async () => { + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + function seed(rowsPerTable: { big: number; small: number }): void { + const db = new Database(dbPath); + try { + db.run("CREATE TABLE big (id INTEGER PRIMARY KEY, v TEXT NOT NULL)"); + db.run("CREATE TABLE small (id INTEGER PRIMARY KEY)"); + const bigStmt = db.prepare("INSERT INTO big (v) VALUES (?)"); + for (let i = 0; i < rowsPerTable.big; i++) bigStmt.run("x"); + const smallStmt = db.prepare("INSERT INTO small DEFAULT VALUES"); + for (let i = 0; i < rowsPerTable.small; i++) smallStmt.run(); + } finally { + db.close(); + } + } + + function analyze(): void { + const db = new Database(dbPath); + try { + db.run("ANALYZE"); + } finally { + db.close(); + } + } + + it("counts small tables exactly", () => { + seed({ big: 10, small: 2 }); + const db = new Database(dbPath, { readonly: true }); + try { + const rendered = renderTableList(listTables(db, { probeCap: 100 })); + expect(rendered).toContain("big (10 rows)"); + expect(rendered).toContain("small (2 rows)"); + } finally { + db.close(); + } + }); + + it("reports the planner estimate for tables larger than the probe cap", () => { + seed({ big: 10, small: 2 }); + analyze(); + const db = new Database(dbPath, { readonly: true }); + try { + // probeCap=5: big (estimate 10) exceeds it and is reported as an estimate + // without scanning; small (estimate 2) is counted exactly. + const rendered = renderTableList(listTables(db, { probeCap: 5 })); + expect(rendered).toContain("big (~10 rows)"); + expect(rendered).toContain("small (2 rows)"); + } finally { + db.close(); + } + }); + + it("reports a lower bound when an unanalyzed table exceeds the probe cap", () => { + seed({ big: 10, small: 2 }); + const db = new Database(dbPath, { readonly: true }); + try { + // No ANALYZE, so no estimate exists; the bounded probe stops at the cap + // and reports a lower bound instead of scanning the whole table. + const rendered = renderTableList(listTables(db, { probeCap: 3 })); + expect(rendered).toContain("big (3+ rows)"); + expect(rendered).toContain("small (2 rows)"); + } finally { + db.close(); + } + }); +});