perf: optimized markdown rendering speed and device doc schemas
- Render device doc parameter schemas as TypeScript types instead of raw JSON schema dumps. - Optimize marked streaming block rules and add pre-gates to reduce CPU overhead. - Increase the markdown render cache entry budget from 32 KiB to 256 KiB.
This commit is contained in:
@@ -2,6 +2,11 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
|
||||
- Usage report filtering in the auth-broker remote store is memoized per (reports, snapshot) with a precomputed per-provider OAuth credential map, replacing an O(reports × credentials) scan on every credential-selection and status refresh
|
||||
- Cursor and Devin Connect-frame readers no longer copy every stream chunk through `Buffer.concat` when the pending buffer is empty
|
||||
|
||||
## [17.1.6] - 2026-07-27
|
||||
|
||||
### Added
|
||||
|
||||
@@ -7,6 +7,12 @@
|
||||
- Added a parser for macOS `sample`(1) call-tree reports to the read tool: `*.sample.txt` reads now return a compact bottleneck summary — per-thread hot paths with on-CPU sample counts (blocked syscall time excluded), demangled Rust v0/legacy symbols, flattened direct recursion, merged call-site siblings, idle-thread classification, and a process-wide top-functions-by-self-samples table. `:raw` still reads the original report, and files that merely carry the extension fall back to plain text.
|
||||
- Added V8 `.cpuprofile` support to the read tool (Node/Bun `--cpu-prof`, Chrome DevTools, CDP `Profiler.stop` output): reads now return a compact bottleneck summary — hot-path call tree with on-CPU milliseconds (`(idle)` time excluded), collapsed pass-through chains, flattened direct recursion, shortened file URLs, and a top-functions-by-self-time table. `:raw` still reads the original JSON, and files that merely carry the extension fall back to plain text.
|
||||
|
||||
### Changed
|
||||
|
||||
- Session listing now caches parsed headers keyed on file stat identity (mtime + size), so repeated resume-picker opens and startup scans re-read only changed session files
|
||||
- Reduced per-keystroke editor dispatch overhead: keybinding resolution happens once per input chunk and the per-action interception chain is gated behind a single canonical-key set probe
|
||||
- `xd://` device docs now render the parameter schema as a comment-annotated TypeScript type (via `jsonSchemaToTypeScript`, the same renderer the in-band tool inventory uses) instead of a raw JSON Schema dump, shrinking system-prompt device sections while keeping descriptions inline.
|
||||
|
||||
## [17.1.6] - 2026-07-27
|
||||
|
||||
### Added
|
||||
|
||||
@@ -24,7 +24,7 @@
|
||||
* the wrapped tool's own renderer with the decoded inner args.
|
||||
*/
|
||||
import type { AgentToolContext, AgentToolResult, AgentToolUpdateCallback, ToolLoadMode } from "@oh-my-pi/pi-agent-core";
|
||||
import { type Tool as AiTool, toolWireSchema, validateToolArguments } from "@oh-my-pi/pi-ai";
|
||||
import { type Tool as AiTool, jsonSchemaToTypeScript, toolWireSchema, validateToolArguments } from "@oh-my-pi/pi-ai";
|
||||
import { type Component, Container, Text } from "@oh-my-pi/pi-tui";
|
||||
import { parseStreamingJson } from "@oh-my-pi/pi-utils";
|
||||
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
|
||||
@@ -107,7 +107,7 @@ function schemaDeclaresIntentField(schema: unknown): boolean {
|
||||
}
|
||||
|
||||
function renderDocs(inst: Tool, heading = "#", descriptionCap?: number): string {
|
||||
const schema = JSON.stringify(toolWireSchema(inst as AiTool), null, 1);
|
||||
const schema = jsonSchemaToTypeScript(toolWireSchema(inst as AiTool));
|
||||
let description = inst.description ?? "";
|
||||
if (descriptionCap !== undefined && description.length > descriptionCap) {
|
||||
description = `${description.slice(0, descriptionCap).trimEnd()}… (full docs: read ${XD_URL_PREFIX}${inst.name})`;
|
||||
@@ -118,8 +118,8 @@ function renderDocs(inst: Tool, heading = "#", descriptionCap?: number): string
|
||||
description,
|
||||
"",
|
||||
`${heading}# Schema`,
|
||||
"```json",
|
||||
schema,
|
||||
"```ts",
|
||||
`type Args = ${schema};`,
|
||||
"```",
|
||||
`Execute by writing JSON to ${XD_URL_PREFIX}${inst.name}.`,
|
||||
].join("\n");
|
||||
|
||||
@@ -97,7 +97,7 @@ export function parseSampleProfile(text: string): SampleProfile | null {
|
||||
const lines = text.split("\n");
|
||||
const analysis = ANALYSIS_RE.exec(lines[0] ?? "");
|
||||
if (!analysis) return null;
|
||||
const callGraphIx = lines.findIndex(line => line === "Call graph:");
|
||||
const callGraphIx = lines.indexOf("Call graph:");
|
||||
if (callGraphIx === -1) return null;
|
||||
|
||||
const header: SampleProfileHeader = {
|
||||
|
||||
@@ -9,7 +9,9 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Optimized markdown URL tokenizer gate, inline math start scan, and autolink scheme scan for performance
|
||||
- Eliminated the dominant markdown streaming CPU cost (73% of a profiled interactive session): marked's GFM `url` tokenizer and `lheading` rule are now gated by O(1)/O(n) charCode pre-checks, the pathological `hr`/`lheading`/`table`/`html` block rules use sticky clones that fail at offset 0 instead of rescanning the source, and the inline math/autolink `start()` scans dropped their regex alternations
|
||||
- Streaming markdown now freezes the stable prefix through provably closed lists instead of re-lexing everything after the last non-list block on every delta
|
||||
- Raised the markdown render cache entry budget (32 KiB → 256 KiB) so large messages — exactly the expensive renders — are cacheable
|
||||
- Deduplicated terminal cursor-visibility writes to skip redundant escape sequences
|
||||
|
||||
## [17.1.6] - 2026-07-27
|
||||
|
||||
@@ -751,16 +751,78 @@ export function urlTokenPossible(src: string): boolean {
|
||||
return src.charCodeAt(i) === 64 /* @ */;
|
||||
}
|
||||
|
||||
// Setext-underline pre-gate for marked's `lheading` rule. The rule's lazy body
|
||||
// `((?:.|\n(?!<block-start>))+?)` re-runs its block-start lookahead while
|
||||
// expanding character by character, so even a FAILING attempt at offset 0
|
||||
// costs O(len × lookahead) — ~26µs per 200-char list-item body, and marked's
|
||||
// list tokenizer block-tokenizes every item's content (47.8% of a streaming
|
||||
// bench profile). A match REQUIRES the setext underline `\n {0,3}(=+|-+)`
|
||||
// somewhere in src, so this O(n) charCode scan never rejects a src the
|
||||
// built-in rule would match; single-line srcs (every tight list item) reject
|
||||
// on the first indexOf.
|
||||
function lheadingPossible(src: string): boolean {
|
||||
let i = src.indexOf("\n");
|
||||
while (i !== -1) {
|
||||
let j = i + 1;
|
||||
const limit = j + 3; // underline allows up to 3 leading spaces
|
||||
while (j < limit && src.charCodeAt(j) === 0x20 /* space */) j++;
|
||||
const c = src.charCodeAt(j); // NaN past the end fails both comparisons
|
||||
if (c === 0x3d /* = */ || c === 0x2d /* - */) return true;
|
||||
i = src.indexOf("\n", j);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
markdownParser.use({
|
||||
tokenizer: {
|
||||
// `false` → marked falls back to the built-in tokenizer;
|
||||
// `undefined` → no token here, built-in never runs.
|
||||
url(src: string): Tokens.Link | undefined | false {
|
||||
// `false` → marked falls back to the built-in `url` tokenizer;
|
||||
// `undefined` → no url token here, built-in never runs.
|
||||
return urlTokenPossible(src) ? false : undefined;
|
||||
},
|
||||
lheading(src: string): Tokens.Heading | undefined | false {
|
||||
return lheadingPossible(src) ? false : undefined;
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Sticky clones of marked's pathological block rules
|
||||
// ---------------------------------------------------------------------------
|
||||
// Bun's (JSC) regex engine skips the start-anchor fast-fail for several of
|
||||
// marked's `^`-anchored block rules — `hr`, `lheading`, `table` and `html` are
|
||||
// anchored alternations of quantified branches, and a failing `exec`/`test`
|
||||
// rescans the entire remaining source instead of stopping after offset 0.
|
||||
// marked's list tokenizer runs `hr.test` and `lheading` per list line against
|
||||
// the remaining source, so lexing a long list is quadratic (66% of a streaming
|
||||
// bench profile sat in these two regexes). A sticky (`y`) clone with
|
||||
// `lastIndex` pinned to 0 attempts the match at offset 0 only.
|
||||
//
|
||||
// Equivalence: for a flagless rule whose source is `^`-anchored, a sticky
|
||||
// clone at `lastIndex = 0` matches exactly when the original matches (same
|
||||
// match object, same captures) — `^` already restricted matches to offset 0
|
||||
// (no `m` flag), and stickiness only removes the futile later attempts. The
|
||||
// flags/anchor guard below skips any rule a future marked version changes.
|
||||
class AnchoredAtZero extends RegExp {
|
||||
exec(str: string): RegExpExecArray | null {
|
||||
this.lastIndex = 0; // sticky matches set lastIndex; rules are shared
|
||||
return super.exec(str);
|
||||
}
|
||||
test(str: string): boolean {
|
||||
this.lastIndex = 0;
|
||||
return super.test(str);
|
||||
}
|
||||
}
|
||||
|
||||
for (const table of [Lexer.rules.block.normal, Lexer.rules.block.gfm]) {
|
||||
for (const name of ["hr", "lheading", "table", "html"] as const) {
|
||||
const rule = table[name];
|
||||
if (rule.flags === "" && rule.source.startsWith("^")) {
|
||||
table[name] = new AnchoredAtZero(rule.source, "y");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Module-level LRU render cache
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user