feat: added advisory transcript formatting and one-shot benchmark metrics

- Introduced advisory note output as `<advisory>` tags with optional severity and guidance.
- Updated session transcript formatting to `### Session update` and inline watched role labels.
- Added shared `escapeXmlText` utility and escaped XML-sensitive text in advisor outputs.
- Added one-shot success run token metrics and one-shot statistics reporting.
This commit is contained in:
can1357
2026-06-16 18:34:50 +02:00
parent 3c0c6daab8
commit 5bce7ed6df
15 changed files with 301 additions and 58 deletions
+4 -2
View File
@@ -85,10 +85,12 @@ The `advise` tool accepts one note and an optional severity:
| `concern` | Interrupting steering message. | Material risk, likely wrong direction, missing constraint, hallucinated API. |
| `blocker` | Interrupting steering message. | Continuing would clearly waste work or produce broken output. |
Interrupting advice is sent through the steering channel and can abort in-flight tools at the next steering boundary. Non-interrupting notes are batched into one custom `advisor` transcript card with this prefix:
Interrupting advice is sent through the steering channel and can abort in-flight tools at the next steering boundary. Each note (interrupting or batched) is rendered into the primary transcript as an `<advisory>` element — severity rides a `severity` attribute, and a `guidance` attribute carries the "weigh, don't blindly obey" framing (the primary agent's system prompt never mentions advisories, so the tag is its only cue). Note bodies are XML-escaped so advice containing `<`, `>`, or `&` can't break the wrapper:
```text
Advisor (a senior reviewer watching your work — weigh it, don't blindly obey):
<advisory severity="concern" guidance="weigh, don't blindly obey">
note text
</advisory>
```
When you deliberately interrupt the agent (Esc, or a cancel from collab, ACP, RPC, the SDK, or an extension), the advisor stops auto-resuming it. An interrupting `concern`/`blocker` raised while the run is stopped is recorded as a visible advisor card instead of restarting the turn, and a concern already in flight when you interrupt is preserved the same way rather than driving a surprise resume. The advice re-enters context the next time you resume — a new message, the `.`/`c` continue shortcut, or a steer/follow-up. A normal yield is unaffected: the advisor can still steer and resume a run the agent ended on its own.
+6 -1
View File
@@ -1,8 +1,13 @@
# Changelog
## [Unreleased]
### Fixed
### Changed
- Made the watched-session transcript sent to the advisor (and shown by `/advisor dump`) clearer: each turn now opens with a `### Session update` heading; watched-agent roles render as inline `**agent**:` / `**user**:` labels instead of level-2 headings that collided with the advisor's own turns; consecutive same-role messages collapse under one label (the watched agent emits one assistant message per tool call); and batched updates are joined by a blank line rather than a `---` rule.
- Changed the compact transcript tool-intent prefix (`history://`, `/dump`, `/advisor dump`) from `# ` to `// ` so intent lines read as comments instead of rendering as Markdown H1 headings.
- Changed the advisor advice injected into the primary transcript from a `Advisor (...): - [severity] note` prose block to one `<advisory severity="…" guidance="weigh, don't blindly obey">…</advisory>` element per note, with XML-escaped bodies. (Relocated the shared `escapeXmlText` helper to `@oh-my-pi/pi-utils`.)
### Fixed
- Fixed magic-keyword steering notices (`ultrathink-notice`, `orchestrate-notice`, `workflow-notice`) to be prepended before the related user message so they influence that same turn
- Fixed dequeuing or popping queued user messages to remove their preceding hidden magic-keyword notice companions, preventing orphaned queued notices
- Fixed queued user steers to auto-resume after interrupts even when the transcript tail is a preserved advisor card or other non-conversational custom message
@@ -9,6 +9,7 @@ import {
ADVISOR_READONLY_TOOL_NAMES,
AdviseTool,
type AdvisorAgent,
type AdvisorNote,
AdvisorRuntime,
type AdvisorRuntimeHost,
formatAdvisorBatchContent,
@@ -52,7 +53,7 @@ describe("advisor", () => {
},
scheduleIdleFlush: () => {},
});
yq.register<{ note: string; severity?: "nit" | "concern" | "blocker" }>("advisor", {
yq.register<AdvisorNote>("advisor", {
build: entries =>
entries.length === 0
? null
@@ -62,9 +63,7 @@ describe("advisor", () => {
display: true,
attribution: "agent",
timestamp: Date.now(),
content:
"Advisor (a senior reviewer watching your work — weigh it, don't blindly obey):\n" +
entries.map(e => `- ${e.severity ? `[${e.severity}] ` : ""}${e.note}`).join("\n"),
content: formatAdvisorBatchContent(entries),
} as AgentMessage),
});
@@ -77,8 +76,9 @@ describe("advisor", () => {
expect(msg.role).toBe("custom");
expect(msg.customType).toBe("advisor");
expect(msg.display).toBe(true);
expect(msg.content).toContain("[blocker] second note");
expect(msg.content).toContain("- first note");
expect(msg.content).toContain("second note");
expect(msg.content).toContain('severity="blocker"');
expect(msg.content).toContain("first note");
});
it("skipIdleFlush prevents idle scheduling", () => {
@@ -124,15 +124,21 @@ describe("advisor", () => {
expect(isInterruptingSeverity(undefined)).toBe(false);
});
it("formats a batch with the advisor prefix and severity-tagged bullets", () => {
it("wraps each note in an advisory tag with severity as an attribute and escapes the body", () => {
const content = formatAdvisorBatchContent([
{ note: "first note" },
{ note: "second note", severity: "blocker" },
{ note: "second <note> & more", severity: "blocker" },
]);
const lines = content.split("\n");
expect(lines[0]).toContain("senior reviewer");
expect(lines[1]).toBe("- first note");
expect(lines[2]).toBe("- [blocker] second note");
// No-severity note: bare advisory tag (no severity attribute).
expect(content).toMatch(/<advisory guidance="[^"]*">\nfirst note\n<\/advisory>/);
// Severity rides an attribute, not an inline `[blocker]` tag or a bullet.
expect(content).toMatch(/<advisory severity="blocker" guidance="[^"]*">/);
expect(content).not.toContain("[blocker]");
expect(content).not.toContain("- first note");
// XML-significant characters in the body are escaped so they can't break the tag.
expect(content).toContain("second &lt;note&gt; &amp; more");
// Exactly one severity attribute (only the blocker note carries one).
expect(content.split('severity="').length - 1).toBe(1);
});
});
@@ -279,6 +285,56 @@ describe("advisor", () => {
expect(promptInputs[0]).not.toContain("note");
});
it("renders the watched delta with a heading, watched-role labels, and no inner ## headings", () => {
const promptInputs: string[] = [];
const agent = makeAgent(promptInputs);
const messages: AgentMessage[] = [
{ role: "user", content: "do the thing", timestamp: 1 } as AgentMessage,
{
role: "assistant",
content: [{ type: "toolCall", id: "a", name: "read", arguments: { path: "x.ts" } }],
timestamp: 2,
} as unknown as AgentMessage,
{
role: "toolResult",
toolCallId: "a",
toolName: "read",
content: [{ type: "text", text: "ok" }],
isError: false,
timestamp: 3,
} as AgentMessage,
{
role: "assistant",
content: [{ type: "toolCall", id: "b", name: "search", arguments: { pattern: "y" } }],
timestamp: 4,
} as unknown as AgentMessage,
{
role: "toolResult",
toolCallId: "b",
toolName: "search",
content: [{ type: "text", text: "ok" }],
isError: false,
timestamp: 5,
} as AgentMessage,
];
const host: AdvisorRuntimeHost = {
snapshotMessages: () => messages,
enqueueAdvice: () => {},
};
const runtime = new AdvisorRuntime(agent, host);
runtime.onTurnEnd();
expect(promptInputs).toHaveLength(1);
const prompt = promptInputs[0];
expect(prompt).toContain("### Session update");
expect(prompt).toContain("**user**:");
expect(prompt).toContain("**agent**:");
// Inner role headings would collide with the advisor's own turns in the dump.
expect(prompt).not.toContain("## assistant");
expect(prompt).not.toContain("## user");
// Consecutive assistant tool-call messages collapse under a single label.
expect(prompt.split("**agent**:").length - 1).toBe(1);
});
it("handles compaction shrink without prompting", () => {
const promptInputs: string[] = [];
const agent = makeAgent(promptInputs);
@@ -1,4 +1,5 @@
import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core";
import { escapeXmlText } from "@oh-my-pi/pi-utils";
import { z } from "zod/v4";
import adviseDescription from "../prompts/advisor/advise-tool.md" with { type: "text" };
@@ -33,15 +34,26 @@ export interface AdvisorMessageDetails {
}
/**
* Prose framing prepended to every batched advisor message. Kept here so the
* non-interrupting YieldQueue dispatcher and the interrupting steer path build
* byte-identical content.
* Behavioral framing for the watched agent — advice, not orders. Carried as a
* tag attribute (rather than a prose header) so the rendered agent-facing output
* stays a clean `<advisory>` block. The primary agent's system prompt never
* mentions advisories, so this is its only cue for how to treat them.
*/
const ADVISOR_BATCH_PREFIX = "Advisor (a senior reviewer watching your work — weigh it, don't blindly obey):";
const ADVISOR_GUIDANCE = "weigh, don't blindly obey";
/** Render one advisor card body from a batch of notes (prefix + one bullet per note). */
/**
* Render a batch of advisor notes as the agent-facing message body: one
* `<advisory>` element per note, severity as an attribute. Shared by the
* non-interrupting YieldQueue dispatcher and the interrupting steer path so both
* build byte-identical content.
*/
export function formatAdvisorBatchContent(notes: readonly AdvisorNote[]): string {
return `${ADVISOR_BATCH_PREFIX}\n${notes.map(n => `- ${n.severity ? `[${n.severity}] ` : ""}${n.note}`).join("\n")}`;
return notes
.map(n => {
const severity = n.severity ? ` severity="${n.severity}"` : "";
return `<advisory${severity} guidance="${ADVISOR_GUIDANCE}">\n${escapeXmlText(n.note)}\n</advisory>`;
})
.join("\n");
}
/**
+10 -3
View File
@@ -157,8 +157,13 @@ export class AdvisorRuntime {
.filter(m => !(m.role === "custom" && (m as { customType?: string }).customType === "advisor"));
this.#lastCount = all.length;
if (delta.length === 0) return null;
const md = formatSessionHistoryMarkdown(delta, { includeThinking: true, includeToolIntent: true });
return md.trim() ? md : null;
const md = formatSessionHistoryMarkdown(delta, {
includeThinking: true,
includeToolIntent: true,
watchedRoles: true,
});
if (!md.trim()) return null;
return `### Session update\n\n${md}`;
}
#notifyWaiters(): void {
@@ -182,7 +187,9 @@ export class AdvisorRuntime {
try {
while (!this.disposed && this.#pending.length) {
const popped = this.#pending.splice(0);
const candidateBatch = popped.map(b => b.text).join("\n\n---\n\n");
// Each delta already opens with a `### Session update` heading, so
// join with a blank line rather than a `---` rule.
const candidateBatch = popped.map(b => b.text).join("\n\n");
const turnsCovered = popped.reduce((sum, b) => sum + b.turns, 0);
const incomingTokens = estimateTokens({
role: "user",
+1 -23
View File
@@ -1,4 +1,4 @@
import { prompt, Snowflake } from "@oh-my-pi/pi-utils";
import { escapeXmlText, prompt, Snowflake } from "@oh-my-pi/pi-utils";
import goalBudgetLimitPrompt from "../prompts/goals/goal-budget-limit.md" with { type: "text" };
import goalContinuationPrompt from "../prompts/goals/goal-continuation.md" with { type: "text" };
import goalModeActivePrompt from "../prompts/goals/goal-mode-active.md" with { type: "text" };
@@ -58,28 +58,6 @@ export function remainingTokens(goal: Goal | null | undefined): number | null {
return Math.max(0, goal.tokenBudget - goal.tokensUsed);
}
export function escapeXmlText(input: string): string {
let firstEscapable = -1;
for (let index = 0; index < input.length; index++) {
const char = input.charCodeAt(index);
if (char === 38 || char === 60 || char === 62) {
firstEscapable = index;
break;
}
}
if (firstEscapable === -1) return input;
let output = input.slice(0, firstEscapable);
for (let index = firstEscapable; index < input.length; index++) {
const char = input[index];
if (char === "&") output += "&amp;";
else if (char === "<") output += "&lt;";
else if (char === ">") output += "&gt;";
else output += char;
}
return output;
}
export function renderTrustedObjective(objective: string): string {
return `<objective>\n${escapeXmlText(objective)}\n</objective>`;
}
@@ -26,6 +26,8 @@ export interface HistoryFormatOptions {
includeThinking?: boolean;
/** Render tool intent comment before tool call lines. */
includeToolIntent?: boolean;
/** Render watched-session roles as inline `**agent**:` / `**user**:` labels (collapsing consecutive same-role messages) instead of `## ` headings, so a primary transcript embedded inside an advisor turn stays visually distinct. */
watchedRoles?: boolean;
}
/** Max length of the primary-arg summary inside `→ tool(...)` lines. */
@@ -125,7 +127,7 @@ function toolCallLine(
const intent = includeToolIntent ? args?.[INTENT_FIELD] : undefined;
if (typeof intent === "string" && intent.trim()) {
const formattedIntent = oneLine(intent, 80);
return `# ${formattedIntent}\n${base}`;
return `// ${formattedIntent}\n${base}`;
}
return base;
}
@@ -191,6 +193,11 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History
}
}
const consumed = new Set<string>();
// In watched mode, consecutive same-role messages collapse under one label
// (the watched agent emits one assistant message per tool call, so otherwise
// every call repeats `**agent**:`). Cleared whenever a
// non-role-labeled line is emitted so the next turn re-labels.
let lastWatchedLabel: string | undefined;
for (const msg of typed) {
switch (msg.role) {
@@ -198,7 +205,17 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History
case "developer": {
const text = contentToText(msg.content);
if (!text.trim()) break;
lines.push(`## ${msg.role}`, "", text, "");
if (opts?.watchedRoles) {
const label = `**${msg.role}**:`;
if (lastWatchedLabel === label) {
lines.push(text, "");
} else {
lines.push(label, text, "");
lastWatchedLabel = label;
}
} else {
lines.push(`## ${msg.role}`, "", text, "");
}
break;
}
case "assistant": {
@@ -217,45 +234,62 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History
// redactedThinking elided entirely (no readable text)
}
if (body.length === 0) break;
lines.push("## assistant", "", ...body, "");
if (opts?.watchedRoles) {
const label = "**agent**:";
if (lastWatchedLabel === label) {
lines.push(...body, "");
} else {
lines.push(label, ...body, "");
lastWatchedLabel = label;
}
} else {
lines.push("## assistant", "", ...body, "");
}
break;
}
case "toolResult": {
// Normally consumed by its toolCall; orphans (e.g. truncated history) get their own line.
if (consumed.has(msg.toolCallId)) break;
lines.push(toolCallLine(msg.toolName, undefined, msg, opts?.includeToolIntent), "");
lastWatchedLabel = undefined;
break;
}
case "bashExecution": {
const bashMsg = msg as BashExecutionMessage;
if (bashMsg.excludeFromContext) break;
lines.push(executionLine("bash", bashMsg.command, bashMsg), "");
lastWatchedLabel = undefined;
break;
}
case "pythonExecution": {
const pythonMsg = msg as PythonExecutionMessage;
if (pythonMsg.excludeFromContext) break;
lines.push(executionLine("python", pythonMsg.code, pythonMsg), "");
lastWatchedLabel = undefined;
break;
}
case "custom":
case "hookMessage": {
lines.push(customOneLiner(msg as CustomMessage | HookMessage), "");
lastWatchedLabel = undefined;
break;
}
case "branchSummary": {
const branchMsg = msg as BranchSummaryMessage;
lines.push(`[branch] from ${branchMsg.fromId}: ${oneLine(branchMsg.summary)}`, "");
lastWatchedLabel = undefined;
break;
}
case "compactionSummary": {
const compactMsg = msg as CompactionSummaryMessage;
lines.push(`[compaction] ${oneLine(compactMsg.summary)}`, "");
lastWatchedLabel = undefined;
break;
}
case "fileMention": {
const fileMsg = msg as FileMentionMessage;
lines.push(`[file-mention] ${oneLine(fileMsg.files.map(f => f.path).join(", "))}`, "");
lastWatchedLabel = undefined;
break;
}
}
@@ -1,6 +1,5 @@
import { describe, expect, it } from "bun:test";
import {
escapeXmlText,
GoalRuntime,
type GoalRuntimeHost,
goalTokenDelta,
@@ -8,6 +7,7 @@ import {
renderTrustedObjective,
} from "@oh-my-pi/pi-coding-agent/goals/runtime";
import type { Goal, GoalModeState, GoalRuntimeEvent, GoalTokenUsage } from "@oh-my-pi/pi-coding-agent/goals/state";
import { escapeXmlText } from "@oh-my-pi/pi-utils";
function createUsage(overrides: Partial<GoalTokenUsage> = {}): GoalTokenUsage {
return {
@@ -92,6 +92,14 @@ describe("formatSessionHistoryMarkdown", () => {
expect(output.startsWith("# Spawnling (idle)\n")).toBe(true);
});
it("renders watched roles using bold text rather than level-2 headers when watchedRoles is true", () => {
const output = formatSessionHistoryMarkdown(buildMessages(), { watchedRoles: true });
expect(output).toContain("**user**:");
expect(output).toContain("**agent**:");
expect(output).not.toContain("## user");
expect(output).not.toContain("## assistant");
});
it("renders an orphan toolResult (truncated history) as its own line", () => {
const output = formatSessionHistoryMarkdown([
{
@@ -145,14 +153,14 @@ describe("formatSessionHistoryMarkdown", () => {
];
const outputWithIntent = formatSessionHistoryMarkdown(messages, { includeToolIntent: true });
expect(outputWithIntent).toContain("# reading config file\n→ read(src/config.ts) ⇒ ok · 1 line");
expect(outputWithIntent).toContain("// reading config file\n→ read(src/config.ts) ⇒ ok · 1 line");
// The long intent should be flattened to one line and truncated to 80 characters (including ellipsis).
expect(outputWithIntent).toContain(
"# reading config file with a very very long and descriptive intent that will exce…\n→ read(src/config.ts) ⇒ ok · 1 line",
"// reading config file with a very very long and descriptive intent that will exce…\n→ read(src/config.ts) ⇒ ok · 1 line",
);
const outputWithoutIntent = formatSessionHistoryMarkdown(messages);
expect(outputWithoutIntent).not.toContain("# reading config file");
expect(outputWithoutIntent).not.toContain("// reading config file");
expect(outputWithoutIntent).toContain("→ read(src/config.ts) ⇒ ok · 1 line");
});
});
@@ -518,10 +518,16 @@ async function main(): Promise<void> {
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
);
console.log(
` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
);
console.log(
` Tokens/task (best total): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`,
` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`,
);
console.log(
` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`,
);
console.log(
` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total}`,
);
if (result.summary.ghostRuns > 0) {
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
@@ -563,6 +569,7 @@ class LiveProgress {
#inputTokens: number[] = [];
#outputTokens: number[] = [];
#totalTokens: number[] = [];
#oneShotSuccessTokens: number[] = [];
#lastLineLength = 0;
constructor(totalRuns: number, runsPerTask: number) {
@@ -586,6 +593,9 @@ class LiveProgress {
if (event.result.success) {
this.#success += 1;
}
if (event.result.success && event.runIndex === 0) {
this.#oneShotSuccessTokens.push(event.result.tokens.total);
}
this.#totalInput += event.result.tokens.input;
this.#totalOutput += event.result.tokens.output;
this.#inputTokens.push(event.result.tokens.input);
@@ -689,6 +699,7 @@ class LiveProgress {
console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
console.log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`);
console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
}
@@ -172,7 +172,7 @@ export function generateReport(result: BenchmarkResult): string {
`| **Tool Input Chars** | ${formatNumber(summary.totalToolCalls.totalInputChars)} | ${formatNumber(Math.round(summary.avgToolCallsPerTask.totalInputChars))} |`,
);
lines.push("");
lines.push("### Tokens & Time");
lines.push("### Tokens & Time (Overall)");
lines.push("");
lines.push("| Metric | Total (best) | Avg/Task | Median | P1 | P99 |");
lines.push("|--------|--------------|----------|--------|----|----|");
@@ -190,6 +190,20 @@ export function generateReport(result: BenchmarkResult): string {
);
lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** | — | — | — |`);
lines.push("");
lines.push("### Tokens & Time (One-shot Successes)");
lines.push("");
lines.push("| Metric | Total | Avg/Task | Median | P1 | P99 |");
lines.push("|--------|-------|----------|--------|----|----|");
lines.push(
`| Input Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.input)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.input)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.input)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.input)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.input)} |`,
);
lines.push(
`| Output Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.output)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.output)} |`,
);
lines.push(
`| Total Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.total)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.total)} |`,
);
lines.push("");
if (summary.hashlineEditSubtypes) {
const order = ["set", "set_range", "insert"] as const;
@@ -875,6 +875,18 @@ export interface BenchmarkSummary {
flakyTasks: number;
/** Tasks where every executed non-ghost run succeeded. */
consistentlyPassingTasks: number;
/** Tasks whose first run succeeded. */
successfulOneShotTasks: number;
/** Tokens summed over the first run of each successfully one-shot task. */
totalOneShotSuccessTokens: TokenStats;
/** Average tokens per successfully one-shot task. */
avgOneShotSuccessTokensPerTask: TokenStats;
/** Median tokens across successfully one-shot tasks. */
medianOneShotSuccessTokensPerTask: TokenStats;
/** 1st-percentile tokens across successfully one-shot tasks. */
p1OneShotSuccessTokensPerTask: TokenStats;
/** 99th-percentile tokens across successfully one-shot tasks. */
p99OneShotSuccessTokensPerTask: TokenStats;
/** Tokens summed over the best run of each task. */
totalTokens: TokenStats;
/** Average tokens per task (sum of best runs / number of tasks). */
@@ -1943,8 +1955,31 @@ export function buildBenchmarkResult(params: {
? bestWithMutationIntent.filter(r => r.mutationIntentMatched).length / bestWithMutationIntent.length
: undefined;
const oneShotSuccessRuns = taskResults
.map(t => t.runs.find(r => r.runIndex === 0))
.filter((r): r is TaskRunResult => Boolean(r?.success));
const successfulOneShotTasks = oneShotSuccessRuns.length;
const oneShotDenom = successfulOneShotTasks || 1;
const totalOneShotSuccessTokens: TokenStats = {
input: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.input, 0),
output: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.output, 0),
total: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.total, 0),
};
const oneShotTokenDistribution = summarizeTokenDistribution(oneShotSuccessRuns);
const taskDenom = tasksWithBestRun || 1;
const summary: BenchmarkSummary = {
successfulOneShotTasks,
totalOneShotSuccessTokens,
avgOneShotSuccessTokensPerTask: {
input: Math.round(totalOneShotSuccessTokens.input / oneShotDenom),
output: Math.round(totalOneShotSuccessTokens.output / oneShotDenom),
total: Math.round(totalOneShotSuccessTokens.total / oneShotDenom),
},
medianOneShotSuccessTokensPerTask: oneShotTokenDistribution.median,
p1OneShotSuccessTokensPerTask: oneShotTokenDistribution.p1,
p99OneShotSuccessTokensPerTask: oneShotTokenDistribution.p99,
totalTasks,
totalRuns,
successfulRuns,
@@ -310,6 +310,56 @@ describe("buildBenchmarkResult", () => {
expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, total: 114 });
expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, total: 546 });
});
it("separates token stats for successfully one-shot tasks vs overall", () => {
// Task 1: Succeeded on run 0 (one-shot success). Tokens: 100
// Task 2: Failed on run 0 (150 tokens), succeeded on run 1 (best run, 50 tokens).
// Task 3: Failed on run 0 (200 tokens).
const tasks = [createTask("t1"), createTask("t2"), createTask("t3")];
const resultsByTask = new Map([
["t1", [createRun(0, true, { tokens: { input: 80, output: 20, total: 100 } })]],
[
"t2",
[
createRun(0, false, { tokens: { input: 120, output: 30, total: 150 } }),
createRun(1, true, { tokens: { input: 40, output: 10, total: 50 } }),
],
],
["t3", [createRun(0, false, { tokens: { input: 160, output: 40, total: 200 } })]],
]);
const result = buildBenchmarkResult({
tasks,
config: {
provider: "anthropic",
model: "claude",
runsPerTask: 2,
timeout: 1000,
taskConcurrency: 1,
},
resultsByTask,
startTime: "2026-04-28T00:00:00.000Z",
endTime: "2026-04-28T00:00:01.000Z",
});
const { summary } = result;
// Overall uses best runs:
// t1 best run: run 0 (100 tokens, success)
// t2 best run: run 1 (50 tokens, success)
// t3 best run: run 0 (200 tokens, fail)
// Total overall tokens: 100 + 50 + 200 = 350
expect(summary.totalTokens.total).toBe(350);
expect(summary.avgTokensPerTask.total).toBe(Math.round(350 / 3)); // 117
// Successfully one-shot tasks (run 0 succeeded):
// t1 succeeded on run 0 (100 tokens)
// t2 failed on run 0
// t3 failed on run 0
// Only t1 counts.
expect(summary.successfulOneShotTasks).toBe(1);
expect(summary.totalOneShotSuccessTokens.total).toBe(100);
expect(summary.avgOneShotSuccessTokensPerTask.total).toBe(100);
});
});
describe("writeConversationDump", () => {
+4 -1
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
### Added
- Added `escapeXmlText` utility to escape XML-significant characters `&`, `<`, and `>` in element body text
## [15.13.3] - 2026-06-15
@@ -143,4 +146,4 @@
### Added
- Added an XDG-aware tiny-title model cache directory helper for coding-agent local title models.
- Added an XDG-aware tiny-title model cache directory helper for coding-agent local title models.
+28
View File
@@ -36,3 +36,31 @@ function sanitizeWellFormedText(text: string): string {
CONTROL_RE.lastIndex = 0;
return stripped.replace(CONTROL_RE, "");
}
/**
* Escape the three XML-significant characters (`&`, `<`, `>`) in text destined
* for an XML/markup element body. Allocation-conscious: returns the input
* unchanged (same reference) when nothing needs escaping. Quotes are left as-is
* — use it for element text, not attribute values.
*/
export function escapeXmlText(input: string): string {
let firstEscapable = -1;
for (let index = 0; index < input.length; index++) {
const char = input.charCodeAt(index);
if (char === 38 || char === 60 || char === 62) {
firstEscapable = index;
break;
}
}
if (firstEscapable === -1) return input;
let output = input.slice(0, firstEscapable);
for (let index = firstEscapable; index < input.length; index++) {
const char = input[index];
if (char === "&") output += "&amp;";
else if (char === "<") output += "&lt;";
else if (char === ">") output += "&gt;";
else output += char;
}
return output;
}