chore(benchmark): cleaned benchmark edit parsing and token accounting

- Relaxed `--edit-variant` parsing in `packages/typescript-edit-benchmark/src/index.ts` to accept any string.
- Adjusted token delta calculation in `runner.ts` to subtract estimated system-prompt overhead per assistant turn.
- Captured initial system-prompt tokens in `runSingleTask` and added rough `estimateTokens` helper for corrected accounting.
- Reframed `scripts/rate-edit-tool.py` prompts to target edit-tool behavior and constraints for each fixture type.
This commit is contained in:
can1357
2026-04-25 23:52:08 +02:00
parent 79bad40cfd
commit 60b2fbf1f9
3 changed files with 33 additions and 30 deletions
@@ -74,7 +74,7 @@ Options:
--tasks <ids> Comma-separated task IDs to run (default: all)
--max-tasks <n> Max tasks to sample (default: 80, 0 = all)
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
--edit-variant <v> Edit variant: replace, patch, hashline, chunk, vim, auto (default: auto)
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, chunk, vim, atom, apply_patch), or auto (default: auto)
--edit-fuzzy <bool> Fuzzy matching: true, false, auto (default: auto)
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto (default: auto)
--auto-format Auto-format output files after verify (debug only)
@@ -307,18 +307,8 @@ async function main(): Promise<void> {
tasksToRun = Array.from({ length: maxTasks }, (_, i) => sorted[Math.floor(i * step)]!);
}
const editVariant = values["edit-variant"] as
| "replace"
| "patch"
| "hashline"
| "chunk"
| "vim"
| "auto"
| undefined;
if (editVariant && !["replace", "patch", "hashline", "chunk", "vim", "auto"].includes(editVariant)) {
console.error(`Invalid edit-variant: ${editVariant}. Must be replace, patch, hashline, chunk, vim, or auto.`);
process.exit(1);
}
const rawEditVariant = values["edit-variant"] as string | undefined;
const editVariant = rawEditVariant === "" ? undefined : rawEditVariant;
let editFuzzy: boolean | "auto" | undefined;
if (values["edit-fuzzy"] !== undefined) {
@@ -40,7 +40,7 @@ interface BenchmarkClient {
onEvent(listener: (event: { type: string; [key: string]: unknown }) => void): () => void;
prompt(text: string): Promise<void>;
followUp(text: string): Promise<void>;
getSessionStats(): Promise<{ tokens: { input: number; output: number; total: number } }>;
getSessionStats(): Promise<{ tokens: { input: number; output: number; total: number }; assistantMessages: number }>;
getLastAssistantText(): Promise<string | null>;
getMessages(): Promise<AgentMessage[]>;
getState(): Promise<ConversationDumpSessionState>;
@@ -70,7 +70,7 @@ export interface BenchmarkConfig {
requireReadToolCall?: boolean;
noEditRequired?: boolean;
autoFormat?: boolean;
editVariant?: "replace" | "patch" | "hashline" | "chunk" | "vim" | "auto";
editVariant?: string;
editFuzzy?: boolean | "auto";
editFuzzyThreshold?: number | "auto";
guided?: boolean;
@@ -961,6 +961,9 @@ async function runSingleTask(
await client.setThinkingLevel(config.thinkingLevel);
}
const initialState = await client.getState();
const systemPromptTokens = estimateTokens(initialState.systemPrompt ?? "");
const maxAttempts = Math.max(1, Math.floor(config.maxAttempts ?? 1));
const maxTimeoutRetries = config.maxTimeoutRetries ?? 3;
const noOpRetryLimit = config.noOpRetryLimit ?? 2;
@@ -1014,7 +1017,7 @@ async function runSingleTask(
throw err;
}
const statsAfter = await client.getSessionStats();
const attemptTokens = diffTokenStats(statsBefore, statsAfter);
const attemptTokens = diffTokenStats(statsBefore, statsAfter, systemPromptTokens);
tokens = {
input: tokens.input + attemptTokens.input,
output: tokens.output + attemptTokens.output,
@@ -1345,7 +1348,7 @@ async function _runRpcBenchmarkRun(
throw err;
}
const statsAfter = await client.getSessionStats();
const attemptTokens = diffTokenStats(statsBefore, statsAfter);
const attemptTokens = diffTokenStats(statsBefore, statsAfter, 0);
tokens = {
input: tokens.input + attemptTokens.input,
output: tokens.output + attemptTokens.output,
@@ -1771,13 +1774,24 @@ async function collectPromptEvents(
return events;
}
/** Rough token estimate (4 chars per token). Used to subtract system prompt overhead. */
function estimateTokens(text: string): number {
return Math.ceil(text.length / 4);
}
function diffTokenStats(
before: { tokens: { input: number; output: number; total: number } },
after: { tokens: { input: number; output: number; total: number } },
before: { tokens: { input: number; output: number; total: number }; assistantMessages: number },
after: { tokens: { input: number; output: number; total: number }; assistantMessages: number },
systemPromptTokens: number,
): TokenStats {
// The system prompt (and tool definitions) live in cacheRead/cacheWrite, not in `input`.
// `input` already excludes the cached system prompt; only `total` (which sums cache too)
// needs the overhead subtracted, once per LLM call.
const calls = Math.max(0, after.assistantMessages - before.assistantMessages);
const overhead = calls * systemPromptTokens;
const input = Math.max(0, after.tokens.input - before.tokens.input);
const output = Math.max(0, after.tokens.output - before.tokens.output);
const total = Math.max(0, after.tokens.total - before.tokens.total);
const total = Math.max(0, after.tokens.total - before.tokens.total - overhead);
return { input, output, total };
}