refactor: restructured hash API and cache tracking across TypeScript and Rust

- Migrated hash API calls from Bun.hash.xxHash64() to Bun.hash() across TypeScript packages for simplified hash generation.
- Consolidated mermaid cache failure tracking by replacing separate failed Set with null values in cache Map.
- Refactored schema-based child extraction in Rust by introducing promotion_fields tracking and schema_wrapper_child() helper function.
- Updated TypeScript configuration files with reformatted arrays and added compiler options for consistency.
This commit is contained in:
can1357
2026-04-10 14:26:06 +02:00
parent b60c0cc3ce
commit d60ee622a4
17 changed files with 133 additions and 61 deletions
+44 -1
View File
@@ -22,6 +22,7 @@ const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[
];
const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"];
const PROMOTION_FIELD_PRIORITY: &[&str] = &["definition", "declaration", "item", "member"];
#[derive(Clone, Copy)]
struct GrammarSpec {
@@ -40,7 +41,9 @@ struct RawTypeRef {
#[derive(Deserialize)]
struct RawFieldSpec {
#[serde(default)]
types: Vec<RawTypeRef>,
multiple: bool,
#[serde(default)]
types: Vec<RawTypeRef>,
}
#[derive(Deserialize)]
@@ -61,6 +64,7 @@ struct GeneratedSchema {
struct GeneratedNodeTypeSchema {
identifier_fields: Vec<String>,
body_fields: Vec<String>,
promotion_fields: Vec<String>,
container_child_kinds: Vec<String>,
is_supertype: bool,
has_structural_children: bool,
@@ -430,6 +434,8 @@ fn build_language_schema(raw_nodes: Vec<RawNodeType>) -> BTreeMap<String, Genera
for (kind, raw) in &raw_by_kind {
let identifier_fields = pick_priority_fields(raw.fields.as_ref(), IDENTIFIER_FIELD_PRIORITY);
let body_fields = pick_priority_fields(raw.fields.as_ref(), BODY_FIELD_PRIORITY);
let promotion_fields =
collect_promotion_fields(raw, &structural_state, &identifier_fields, &body_fields);
let container_child_kinds = collect_child_container_kinds(raw, &structural_state);
let is_supertype = is_supertype(raw);
let has_structural_children = structural_state
@@ -438,6 +444,7 @@ fn build_language_schema(raw_nodes: Vec<RawNodeType>) -> BTreeMap<String, Genera
if identifier_fields.is_empty()
&& body_fields.is_empty()
&& promotion_fields.is_empty()
&& container_child_kinds.is_empty()
&& !is_supertype
&& !has_structural_children
@@ -448,6 +455,7 @@ fn build_language_schema(raw_nodes: Vec<RawNodeType>) -> BTreeMap<String, Genera
out.insert(kind.clone(), GeneratedNodeTypeSchema {
identifier_fields,
body_fields,
promotion_fields,
container_child_kinds,
is_supertype,
has_structural_children,
@@ -502,6 +510,41 @@ fn collect_child_container_kinds(
kinds.into_iter().collect()
}
fn collect_promotion_fields(
raw: &RawNodeType,
structural_state: &HashMap<String, StructuralState>,
identifier_fields: &[String],
body_fields: &[String],
) -> Vec<String> {
let Some(fields) = raw.fields.as_ref() else {
return Vec::new();
};
PROMOTION_FIELD_PRIORITY
.iter()
.filter_map(|field_name| {
let spec = fields.get(*field_name)?;
if spec.multiple
|| identifier_fields.iter().any(|field| field == field_name)
|| body_fields.iter().any(|field| field == field_name)
{
return None;
}
let has_structural_type = spec.types.iter().any(|field_type| {
field_type.named
&& field_type
.kind
.as_deref()
.and_then(|kind| structural_state.get(kind))
.copied()
.is_some_and(StructuralState::is_structural)
});
has_structural_type.then(|| (*field_name).to_string())
})
.collect()
}
#[derive(Clone, Copy, Default)]
struct StructuralState {
is_structural: bool,
+22 -5
View File
@@ -13,6 +13,7 @@ use super::{
},
defaults,
kind::ChunkKind,
schema,
};
use crate::chunk::types::ChunkNode;
@@ -240,6 +241,10 @@ pub fn first_wrapper_content_child<'tree>(
classifier: &dyn LangClassifier,
node: Node<'tree>,
) -> Option<Node<'tree>> {
if let Some(child) = schema_wrapper_child(node) {
return Some(child);
}
let overrides = structural_overrides(classifier);
named_children(node)
.into_iter()
@@ -340,6 +345,13 @@ fn promotable_wrapper_child<'tree>(
node: Node<'tree>,
source: &str,
) -> Option<(Node<'tree>, RawChunkCandidate<'tree>)> {
if let Some(child) = schema_wrapper_child(node) {
let candidate = classify_with_defaults(classifier, context, child, source);
if is_promotable_wrapper_candidate(child, &candidate) {
return Some((child, candidate));
}
}
let overrides = structural_overrides(classifier);
let mut promoted = named_children(node).into_iter().filter_map(|child| {
if is_wrapper_metadata_child(child, classifier, overrides) {
@@ -372,11 +384,6 @@ fn is_wrapper_metadata_child(
|| is_absorbable_attribute(kind)
|| overrides.is_absorbable_attr(kind)
|| classifier.is_absorbable_attr(kind)
|| matches!(kind, "annotation" | "annotations" | "decorator" | "modifier" | "modifiers")
|| kind.ends_with("_annotation")
|| kind.ends_with("_attribute")
|| kind.ends_with("_decorator")
|| kind.ends_with("_modifier")
}
fn is_promotable_wrapper_candidate(node: Node<'_>, candidate: &RawChunkCandidate<'_>) -> bool {
@@ -391,6 +398,16 @@ fn is_promotable_wrapper_candidate(node: Node<'_>, candidate: &RawChunkCandidate
|| node.kind().ends_with("_declaration")
}
fn schema_wrapper_child(node: Node<'_>) -> Option<Node<'_>> {
let schema = schema::schema_for_current(node.kind())?;
for field in &schema.promotion_fields {
if let Some(child) = node.child_by_field_name(field) {
return Some(child);
}
}
None
}
/// Resolve a [`LangClassifier`] for the given language.
pub fn classifier_for(lang: &str) -> &'static dyn LangClassifier {
match lang {
+19
View File
@@ -11,6 +11,7 @@ struct GeneratedSchema {
pub struct NodeTypeSchema {
pub identifier_fields: Vec<String>,
pub body_fields: Vec<String>,
pub promotion_fields: Vec<String>,
pub container_child_kinds: Vec<String>,
pub is_supertype: bool,
pub has_structural_children: bool,
@@ -84,6 +85,7 @@ mod tests {
.expect("python function_definition schema should exist");
assert_eq!(schema.identifier_fields, vec!["name".to_string()]);
assert_eq!(schema.body_fields, vec!["body".to_string()]);
assert!(schema.promotion_fields.is_empty());
assert!(schema.is_structural());
}
@@ -102,6 +104,23 @@ mod tests {
assert!(schema.has_structural_children);
}
#[test]
fn wrapper_schemas_preserve_promotable_definition_fields() {
let python = schema_for("python", "decorated_definition")
.expect("python decorated_definition schema should exist");
assert_eq!(python.promotion_fields, vec!["definition".to_string()]);
let typescript = schema_for("typescript", "export_statement")
.expect("typescript export_statement schema should exist");
assert!(
typescript
.promotion_fields
.iter()
.any(|field| field == "declaration"),
"export_statement should preserve declaration field for promotion"
);
}
#[test]
fn generated_schema_covers_expected_languages() {
for language in ["python", "nix", "toml", "typescript", "rust", "yaml"] {
@@ -2134,7 +2134,7 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
if (!msgId) {
msgId = `msg_${msgIndex}`;
} else if (msgId.length > 64) {
msgId = `msg_${Bun.hash.xxHash64(msgId).toString(36)}`;
msgId = `msg_${Bun.hash(msgId).toString(36)}`;
}
outputItems.push({
type: "message",
@@ -805,9 +805,7 @@ export function convertMessages(
const generateFallbackToolCallId = (seed: string): string => {
generatedToolCallIdCounter += 1;
const hash = Bun.hash
.xxHash64(`${model.provider}:${model.id}:${seed}:${generatedToolCallIdCounter}`)
.toString(36);
const hash = Bun.hash(`${model.provider}:${model.id}:${seed}:${generatedToolCallIdCounter}`).toString(36);
return `call_${hash}`;
};
@@ -144,7 +144,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
if (!msgId) {
msgId = `msg_${msgIndex}`;
} else if (msgId.length > 64) {
msgId = `msg_${Bun.hash.xxHash64(msgId).toString(36)}`;
msgId = `msg_${Bun.hash(msgId).toString(36)}`;
}
outputItems.push({
type: "message",
+3 -3
View File
@@ -38,7 +38,7 @@ export function normalizeResponsesToolCallId(id: string): { callId: string; item
const normalizedItemId = normalizeResponsesItemId(itemId);
return { callId: normalizedCallId, itemId: normalizedItemId };
}
const hash = Bun.hash.xxHash64(id).toString(36);
const hash = Bun.hash(id).toString(36);
const normalizedCallId = id.startsWith("call_") ? truncateResponseItemId(id, "call") : `call_${hash}`;
return { callId: normalizedCallId, itemId: `fc_${hash}` };
}
@@ -51,7 +51,7 @@ function getIdPrefix(id: string, fallback: string): string {
function normalizeResponsesItemId(itemId: string): string {
const prefix = getIdPrefix(itemId, "fc");
if (prefix !== "fc" && prefix !== "fcr") {
return `fc_${Bun.hash.xxHash64(itemId).toString(36)}`;
return `fc_${Bun.hash(itemId).toString(36)}`;
}
return truncateResponseItemId(itemId, prefix);
}
@@ -62,7 +62,7 @@ function normalizeResponsesItemId(itemId: string): string {
*/
export function truncateResponseItemId(id: string, prefix: string): string {
if (id.length <= 64) return id;
return `${prefix}_${Bun.hash.xxHash64(id).toString(36)}`;
return `${prefix}_${Bun.hash(id).toString(36)}`;
}
export function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Array<Record<string, unknown>>): ResponseInput {
@@ -1,7 +1,6 @@
import { extractMermaidBlocks, logger, renderMermaidAsciiSafe } from "@oh-my-pi/pi-utils";
const cache = new Map<bigint, string>();
const failed = new Set<bigint>();
const cache = new Map<bigint | number, string | null>();
let onRenderNeeded: (() => void) | null = null;
@@ -16,7 +15,7 @@ export function setMermaidRenderCallback(callback: (() => void) | null): void {
* Get a pre-rendered mermaid ASCII diagram by hash.
* Returns null if not cached or rendering failed.
*/
export function getMermaidAscii(hash: bigint): string | null {
export function getMermaidAscii(hash: bigint | number): string | null {
return cache.get(hash) ?? null;
}
@@ -31,14 +30,14 @@ export function prerenderMermaid(markdown: string): void {
let hasNew = false;
for (const { source, hash } of blocks) {
if (cache.has(hash) || failed.has(hash)) continue;
if (cache.has(hash)) continue;
const ascii = renderMermaidAsciiSafe(source);
if (ascii) {
cache.set(hash, ascii);
hasNew = true;
} else {
failed.add(hash);
cache.set(hash, null);
}
}
@@ -58,7 +57,7 @@ export function prerenderMermaid(markdown: string): void {
*/
export function hasPendingMermaid(markdown: string): boolean {
const blocks = extractMermaidBlocks(markdown);
return blocks.some(({ hash }) => !cache.has(hash) && !failed.has(hash));
return blocks.some(({ hash }) => !cache.has(hash));
}
/**
@@ -66,5 +65,4 @@ export function hasPendingMermaid(markdown: string): boolean {
*/
export function clearMermaidCache(): void {
cache.clear();
failed.clear();
}
@@ -761,7 +761,7 @@ function buildOpenAiNativeHistory(
if (!msgId) {
msgId = `msg_${msgIndex}`;
} else if (msgId.length > 64) {
msgId = `msg_${Bun.hash.xxHash64(msgId).toString(36)}`;
msgId = `msg_${Bun.hash(msgId).toString(36)}`;
}
input.push({
type: "message",
+3
View File
@@ -1,8 +1,11 @@
# Changelog
## [Unreleased]
### Changed
- Updated hash computation to use `Bun.hash()` instead of `Bun.hash.xxHash64()`, which may return `number` in addition to `bigint`
- Simplified cache key computation in Box component by removing intermediate hash updates and consolidating hash operations
- Wrapped native text utility functions (`sliceWithWidth`, `truncateToWidth`, `wrapTextWithAnsi`, `extractSegments`) to automatically pass the current default tab width, simplifying the API for consumers
- Added `getIndentationNoescape` wrapper that uses `process.cwd()` as the project root for relative file paths
- Re-export `getDefaultTabWidth`, `getIndentation`, and `setDefaultTabWidth` from `@oh-my-pi/pi-utils`; native text helpers still receive tab width via wrappers that read the JS default
+8 -8
View File
@@ -2,7 +2,7 @@ import type { Component } from "../tui";
import { applyBackgroundToLine, padding, visibleWidth } from "../utils";
type Cache = {
key: bigint;
key: bigint | number;
result: string[];
};
@@ -52,20 +52,20 @@ export class Box implements Component {
}
static #tmp = new Uint32Array(2);
#computeCacheKey(width: number, childLines: string[], bgSample: string | undefined): bigint {
#computeCacheKey(width: number, childLines: string[], bgSample: string | undefined): bigint | number {
Box.#tmp[0] = width;
Box.#tmp[1] = childLines.length;
let h = Bun.hash.xxHash64(Box.#tmp);
let h = Bun.hash(Box.#tmp);
for (const line of childLines) {
Box.#tmp[0] = line.length;
h = Bun.hash.xxHash64(Box.#tmp, h);
h = Bun.hash.xxHash64(line, h);
h = Bun.hash(line, h);
}
if (bgSample) {
h = Bun.hash(bgSample, h);
}
h = Bun.hash.xxHash64(bgSample ?? "", h);
return h;
}
#matchCache(cacheKey: bigint): boolean {
#matchCache(cacheKey: bigint | number): boolean {
return this.#cached?.key === cacheKey;
}
+2 -3
View File
@@ -45,10 +45,9 @@ export interface MarkdownTheme {
highlightCode?: (code: string, lang?: string) => string[];
/**
* Lookup a pre-rendered mermaid ASCII rendering by source hash.
* Hash is computed as `Bun.hash.xxHash64(source.trim())`.
* Return null to fall back to fenced code rendering.
*/
getMermaidAscii?: (sourceHash: bigint) => string | null;
getMermaidAscii?: (sourceHash: bigint | number) => string | null;
symbols: SymbolTheme;
}
@@ -325,7 +324,7 @@ export class Markdown implements Component {
case "code": {
// Handle mermaid diagrams with ASCII rendering when available
if (token.lang === "mermaid" && this.#theme.getMermaidAscii) {
const hash = Bun.hash.xxHash64(token.text.trim());
const hash = Bun.hash(token.text.trim());
const ascii = this.#theme.getMermaidAscii(hash);
if (ascii) {
@@ -672,7 +672,7 @@ function buildBenchmarkProviderSessionId(params: {
`system:${buildBenchmarkSystemPrompt({ multiFile: params.multiFile, config: params.config })}`,
`initial:${buildInitialBenchmarkPrompt({ taskPrompt: params.task.prompt, guidedContext: params.initialGuidedContext })}`,
].join("\n");
return `reb_${Bun.hash.xxHash64(keyMaterial).toString(36)}`;
return `reb_${Bun.hash(keyMaterial).toString(36)}`;
}
async function prepareBenchmarkSessionSetup(params: {
+3 -3
View File
@@ -17,13 +17,13 @@ export function renderMermaidAsciiSafe(source: string, options?: AsciiRenderOpti
/**
* Extract mermaid code blocks from markdown text.
*/
export function extractMermaidBlocks(markdown: string): { source: string; hash: bigint }[] {
const blocks: { source: string; hash: bigint }[] = [];
export function extractMermaidBlocks(markdown: string): { source: string; hash: bigint | number }[] {
const blocks: { source: string; hash: bigint | number }[] = [];
const regex = /```mermaid\s*\n([\s\S]*?)```/g;
for (let match = regex.exec(markdown); match !== null; match = regex.exec(markdown)) {
const source = match[1].trim();
const hash = Bun.hash.xxHash64(source);
const hash = Bun.hash(source);
blocks.push({ source, hash });
}
+8 -6
View File
@@ -12,6 +12,8 @@ import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
type CacheKey = string | bigint | number;
// Tools shipped by Xcode / Command Line Tools that callers actually look up.
// Keeps the set small so darwinWhich can fast-reject non-Xcode commands without
// touching the filesystem. Only needs entries for binaries that live *exclusively*
@@ -146,7 +148,7 @@ function getMacosToolPaths(): Map<string, string> {
}
// Map: cache key -> resolved binary path or null (not found)
const toolCache = new Map<string | bigint, string | null>();
const toolCache = new Map<CacheKey, string | null>();
/**
* Cache policy for which lookups.
@@ -194,12 +196,12 @@ function darwinWhich(command: string, _options?: Bun.WhichOptions): string | nul
export const whichFresh = os.platform() === "darwin" ? darwinWhich : Bun.which;
// Derive stable cache key from command and lookup options
function cacheKey(command: string, options?: Bun.WhichOptions): string | bigint {
function cacheKey(command: string, options?: Bun.WhichOptions): CacheKey {
if (!options) return command;
if (!options.cwd && !options.PATH) return command;
let h = Bun.hash.xxHash64(command);
if (options.cwd) h = Bun.hash.xxHash64(options.cwd, h);
if (options.PATH) h = Bun.hash.xxHash64(options.PATH, h);
let h = Bun.hash(command);
if (options.cwd) h = Bun.hash(options.cwd, h);
if (options.PATH) h = Bun.hash(options.PATH, h);
return h;
}
@@ -212,7 +214,7 @@ function cacheKey(command: string, options?: Bun.WhichOptions): string | bigint
*/
export function $which(command: string, options?: WhichOptions): string | null {
const cachePolicy = options?.cache ?? WhichCachePolicy.Cached;
let key: string | bigint | undefined;
let key: CacheKey | undefined;
if (cachePolicy !== WhichCachePolicy.Bypass) {
key = cacheKey(command, options);
+4 -12
View File
@@ -2,10 +2,7 @@
"compilerOptions": {
"target": "ES2024",
"module": "ESNext",
"lib": [
"ES2024",
"DOM.AsyncIterable"
],
"lib": ["ES2024", "DOM.AsyncIterable"],
"moduleResolution": "Bundler",
"moduleDetection": "force",
"strict": true,
@@ -13,6 +10,7 @@
"allowArbitraryExtensions": true,
"verbatimModuleSyntax": true,
"noEmit": true,
"emitDeclarationOnly": false,
"declaration": true,
"declarationMap": true,
"sourceMap": true,
@@ -24,14 +22,8 @@
"experimentalDecorators": true,
"emitDecoratorMetadata": true,
"useDefineForClassFields": false,
"types": [
"bun",
"assets"
],
"typeRoots": [
"./types",
"./node_modules/@types"
],
"types": ["bun", "assets"],
"typeRoots": ["./types", "./node_modules/@types"],
"assumeChangesOnlyAffectDirectDependencies": true
}
}
+7 -6
View File
@@ -1,9 +1,10 @@
{
"extends": "./tsconfig.base.json",
"include": [
"scripts"
],
"exclude": [
"node_modules"
]
"compilerOptions": {
"composite": true,
"noEmit": true,
"emitDeclarationOnly": false
},
"include": ["scripts"],
"exclude": ["node_modules"]
}