diff --git a/packages/typescript-edit-benchmark/src/index.ts b/packages/typescript-edit-benchmark/src/index.ts index 4c5e8919c..0eb202dce 100644 --- a/packages/typescript-edit-benchmark/src/index.ts +++ b/packages/typescript-edit-benchmark/src/index.ts @@ -74,7 +74,7 @@ Options: --tasks Comma-separated task IDs to run (default: all) --max-tasks Max tasks to sample (default: 80, 0 = all) --fixtures Fixtures directory or .tar.gz archive (default: built-in) - --edit-variant Edit variant: replace, patch, hashline, chunk, vim, auto (default: auto) + --edit-variant Edit variant: any string (e.g. replace, patch, hashline, chunk, vim, atom, apply_patch), or auto (default: auto) --edit-fuzzy Fuzzy matching: true, false, auto (default: auto) --edit-fuzzy-threshold Fuzzy threshold 0-1 or auto (default: auto) --auto-format Auto-format output files after verify (debug only) @@ -307,18 +307,8 @@ async function main(): Promise { tasksToRun = Array.from({ length: maxTasks }, (_, i) => sorted[Math.floor(i * step)]!); } - const editVariant = values["edit-variant"] as - | "replace" - | "patch" - | "hashline" - | "chunk" - | "vim" - | "auto" - | undefined; - if (editVariant && !["replace", "patch", "hashline", "chunk", "vim", "auto"].includes(editVariant)) { - console.error(`Invalid edit-variant: ${editVariant}. Must be replace, patch, hashline, chunk, vim, or auto.`); - process.exit(1); - } + const rawEditVariant = values["edit-variant"] as string | undefined; + const editVariant = rawEditVariant === "" ? undefined : rawEditVariant; let editFuzzy: boolean | "auto" | undefined; if (values["edit-fuzzy"] !== undefined) { diff --git a/packages/typescript-edit-benchmark/src/runner.ts b/packages/typescript-edit-benchmark/src/runner.ts index eeee61680..bae6fe70c 100644 --- a/packages/typescript-edit-benchmark/src/runner.ts +++ b/packages/typescript-edit-benchmark/src/runner.ts @@ -40,7 +40,7 @@ interface BenchmarkClient { onEvent(listener: (event: { type: string; [key: string]: unknown }) => void): () => void; prompt(text: string): Promise; followUp(text: string): Promise; - getSessionStats(): Promise<{ tokens: { input: number; output: number; total: number } }>; + getSessionStats(): Promise<{ tokens: { input: number; output: number; total: number }; assistantMessages: number }>; getLastAssistantText(): Promise; getMessages(): Promise; getState(): Promise; @@ -70,7 +70,7 @@ export interface BenchmarkConfig { requireReadToolCall?: boolean; noEditRequired?: boolean; autoFormat?: boolean; - editVariant?: "replace" | "patch" | "hashline" | "chunk" | "vim" | "auto"; + editVariant?: string; editFuzzy?: boolean | "auto"; editFuzzyThreshold?: number | "auto"; guided?: boolean; @@ -961,6 +961,9 @@ async function runSingleTask( await client.setThinkingLevel(config.thinkingLevel); } + const initialState = await client.getState(); + const systemPromptTokens = estimateTokens(initialState.systemPrompt ?? ""); + const maxAttempts = Math.max(1, Math.floor(config.maxAttempts ?? 1)); const maxTimeoutRetries = config.maxTimeoutRetries ?? 3; const noOpRetryLimit = config.noOpRetryLimit ?? 2; @@ -1014,7 +1017,7 @@ async function runSingleTask( throw err; } const statsAfter = await client.getSessionStats(); - const attemptTokens = diffTokenStats(statsBefore, statsAfter); + const attemptTokens = diffTokenStats(statsBefore, statsAfter, systemPromptTokens); tokens = { input: tokens.input + attemptTokens.input, output: tokens.output + attemptTokens.output, @@ -1345,7 +1348,7 @@ async function _runRpcBenchmarkRun( throw err; } const statsAfter = await client.getSessionStats(); - const attemptTokens = diffTokenStats(statsBefore, statsAfter); + const attemptTokens = diffTokenStats(statsBefore, statsAfter, 0); tokens = { input: tokens.input + attemptTokens.input, output: tokens.output + attemptTokens.output, @@ -1771,13 +1774,24 @@ async function collectPromptEvents( return events; } +/** Rough token estimate (4 chars per token). Used to subtract system prompt overhead. */ +function estimateTokens(text: string): number { + return Math.ceil(text.length / 4); +} + function diffTokenStats( - before: { tokens: { input: number; output: number; total: number } }, - after: { tokens: { input: number; output: number; total: number } }, + before: { tokens: { input: number; output: number; total: number }; assistantMessages: number }, + after: { tokens: { input: number; output: number; total: number }; assistantMessages: number }, + systemPromptTokens: number, ): TokenStats { + // The system prompt (and tool definitions) live in cacheRead/cacheWrite, not in `input`. + // `input` already excludes the cached system prompt; only `total` (which sums cache too) + // needs the overhead subtracted, once per LLM call. + const calls = Math.max(0, after.assistantMessages - before.assistantMessages); + const overhead = calls * systemPromptTokens; const input = Math.max(0, after.tokens.input - before.tokens.input); const output = Math.max(0, after.tokens.output - before.tokens.output); - const total = Math.max(0, after.tokens.total - before.tokens.total); + const total = Math.max(0, after.tokens.total - before.tokens.total - overhead); return { input, output, total }; } diff --git a/scripts/rate-edit-tool.py b/scripts/rate-edit-tool.py index ea158b792..98e1b90bf 100755 --- a/scripts/rate-edit-tool.py +++ b/scripts/rate-edit-tool.py @@ -52,7 +52,6 @@ MODELS = [ "openrouter/moonshotai/kimi-k2.5", "openrouter/anthropic/claude-haiku-4.5", "openrouter/anthropic/claude-sonnet-4.6", - "openrouter/google/gemini-3-flash-preview", "openrouter/z-ai/glm-5-turbo", "openrouter/minimax/minimax-m2.7", ] @@ -61,21 +60,21 @@ ORACLE_MODEL = "openrouter/anthropic/claude-opus-4.6" PROMPT = textwrap.dedent( """\ - You are evaluating the code-reading and code-editing tools on files in this directory. + You are evaluating the **edit** tool on files in this directory. The `read` tool is available so you can inspect file state before and after edits, but it is not under review — do not report on it. {FIXTURE_SURFACE} Work in this order: - 1. Map the surface area. Identify operations, selectors, and addressing modes that actually work. Note differences across file types. + 1. Map the edit surface. Inspect the edit tool schema. Identify every operation and addressing mode the active variant exposes (substring replace, line-anchored ops, structural selectors, append/prepend, etc.). Note any behavior differences you anticipate across file types. - 2. Exercise the supported paths. Read: whole-file, structural chunks, nested members, line ranges, raw source. Edit: replace, insert into containers, insert before/after, delete. On `main.md`, check how addressing differs from code files. + 2. Exercise every supported edit operation against each fixture. Perform the full range of supported mutations — replacing content, inserting above/below a target, deleting, substring rewrites, append/prepend — using only what the schema actually exposes. - 3. Push into awkward cases. Test first/last-child edits, indentation preservation, decorators, docstrings, enum variants, markdown tables, and fenced code blocks. Note whether error messages were clear and actionable. + 3. Push into awkward cases. Probe boundary conditions: first/last line of file, indentation-sensitive blocks (Python), nested members (decorators, methods, enum variants, Go interfaces/generics), markdown tables, fenced code blocks. Note whether error messages were clear and actionable when something went wrong. - 4. Verify after edits. Re-read each file after meaningful edits and confirm no unintended changes. + 4. Verify after edits. Re-read each file after meaningful edits and confirm only the intended lines changed. - Report concrete findings: + Report concrete findings about the edit tool only: - what required workarounds - what was impossible - which errors were clear vs unclear @@ -145,9 +144,9 @@ ORACLE_REVIEW_PROMPT = textwrap.dedent( ).strip() TODOS = [ - "Map the read and edit surface area across every fixture: main.ts, main.rs, main.go, main.py, main.md.", - "Exercise supported read and edit paths on each file with concrete before/after verification.", - "Probe awkward selector, indentation, and boundary cases: decorators, docstrings, enum variants, Go interfaces/generics, markdown tables, fenced blocks.", + "Map the edit tool surface area across every fixture: main.ts, main.rs, main.go, main.py, main.md.", + "Exercise every supported edit operation on each fixture with concrete before/after verification.", + "Probe awkward boundary cases: decorators, docstrings, enum variants, Go interfaces/generics, indentation-sensitive blocks, markdown tables, fenced code blocks.", "Summarize what was awkward, impossible, ambiguous, or under-documented with concrete examples spanning all fixtures.", ]