feat: added advisory transcript formatting and one-shot benchmark metrics
- Introduced advisory note output as `<advisory>` tags with optional severity and guidance. - Updated session transcript formatting to `### Session update` and inline watched role labels. - Added shared `escapeXmlText` utility and escaped XML-sensitive text in advisor outputs. - Added one-shot success run token metrics and one-shot statistics reporting.
This commit is contained in:
@@ -85,10 +85,12 @@ The `advise` tool accepts one note and an optional severity:
|
||||
| `concern` | Interrupting steering message. | Material risk, likely wrong direction, missing constraint, hallucinated API. |
|
||||
| `blocker` | Interrupting steering message. | Continuing would clearly waste work or produce broken output. |
|
||||
|
||||
Interrupting advice is sent through the steering channel and can abort in-flight tools at the next steering boundary. Non-interrupting notes are batched into one custom `advisor` transcript card with this prefix:
|
||||
Interrupting advice is sent through the steering channel and can abort in-flight tools at the next steering boundary. Each note (interrupting or batched) is rendered into the primary transcript as an `<advisory>` element — severity rides a `severity` attribute, and a `guidance` attribute carries the "weigh, don't blindly obey" framing (the primary agent's system prompt never mentions advisories, so the tag is its only cue). Note bodies are XML-escaped so advice containing `<`, `>`, or `&` can't break the wrapper:
|
||||
|
||||
```text
|
||||
Advisor (a senior reviewer watching your work — weigh it, don't blindly obey):
|
||||
<advisory severity="concern" guidance="weigh, don't blindly obey">
|
||||
note text
|
||||
</advisory>
|
||||
```
|
||||
|
||||
When you deliberately interrupt the agent (Esc, or a cancel from collab, ACP, RPC, the SDK, or an extension), the advisor stops auto-resuming it. An interrupting `concern`/`blocker` raised while the run is stopped is recorded as a visible advisor card instead of restarting the turn, and a concern already in flight when you interrupt is preserved the same way rather than driving a surprise resume. The advice re-enters context the next time you resume — a new message, the `.`/`c` continue shortcut, or a steer/follow-up. A normal yield is unaffected: the advisor can still steer and resume a run the agent ended on its own.
|
||||
|
||||
@@ -1,8 +1,13 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Fixed
|
||||
### Changed
|
||||
|
||||
- Made the watched-session transcript sent to the advisor (and shown by `/advisor dump`) clearer: each turn now opens with a `### Session update` heading; watched-agent roles render as inline `**agent**:` / `**user**:` labels instead of level-2 headings that collided with the advisor's own turns; consecutive same-role messages collapse under one label (the watched agent emits one assistant message per tool call); and batched updates are joined by a blank line rather than a `---` rule.
|
||||
- Changed the compact transcript tool-intent prefix (`history://`, `/dump`, `/advisor dump`) from `# ` to `// ` so intent lines read as comments instead of rendering as Markdown H1 headings.
|
||||
- Changed the advisor advice injected into the primary transcript from a `Advisor (...): - [severity] note` prose block to one `<advisory severity="…" guidance="weigh, don't blindly obey">…</advisory>` element per note, with XML-escaped bodies. (Relocated the shared `escapeXmlText` helper to `@oh-my-pi/pi-utils`.)
|
||||
|
||||
### Fixed
|
||||
- Fixed magic-keyword steering notices (`ultrathink-notice`, `orchestrate-notice`, `workflow-notice`) to be prepended before the related user message so they influence that same turn
|
||||
- Fixed dequeuing or popping queued user messages to remove their preceding hidden magic-keyword notice companions, preventing orphaned queued notices
|
||||
- Fixed queued user steers to auto-resume after interrupts even when the transcript tail is a preserved advisor card or other non-conversational custom message
|
||||
|
||||
@@ -9,6 +9,7 @@ import {
|
||||
ADVISOR_READONLY_TOOL_NAMES,
|
||||
AdviseTool,
|
||||
type AdvisorAgent,
|
||||
type AdvisorNote,
|
||||
AdvisorRuntime,
|
||||
type AdvisorRuntimeHost,
|
||||
formatAdvisorBatchContent,
|
||||
@@ -52,7 +53,7 @@ describe("advisor", () => {
|
||||
},
|
||||
scheduleIdleFlush: () => {},
|
||||
});
|
||||
yq.register<{ note: string; severity?: "nit" | "concern" | "blocker" }>("advisor", {
|
||||
yq.register<AdvisorNote>("advisor", {
|
||||
build: entries =>
|
||||
entries.length === 0
|
||||
? null
|
||||
@@ -62,9 +63,7 @@ describe("advisor", () => {
|
||||
display: true,
|
||||
attribution: "agent",
|
||||
timestamp: Date.now(),
|
||||
content:
|
||||
"Advisor (a senior reviewer watching your work — weigh it, don't blindly obey):\n" +
|
||||
entries.map(e => `- ${e.severity ? `[${e.severity}] ` : ""}${e.note}`).join("\n"),
|
||||
content: formatAdvisorBatchContent(entries),
|
||||
} as AgentMessage),
|
||||
});
|
||||
|
||||
@@ -77,8 +76,9 @@ describe("advisor", () => {
|
||||
expect(msg.role).toBe("custom");
|
||||
expect(msg.customType).toBe("advisor");
|
||||
expect(msg.display).toBe(true);
|
||||
expect(msg.content).toContain("[blocker] second note");
|
||||
expect(msg.content).toContain("- first note");
|
||||
expect(msg.content).toContain("second note");
|
||||
expect(msg.content).toContain('severity="blocker"');
|
||||
expect(msg.content).toContain("first note");
|
||||
});
|
||||
|
||||
it("skipIdleFlush prevents idle scheduling", () => {
|
||||
@@ -124,15 +124,21 @@ describe("advisor", () => {
|
||||
expect(isInterruptingSeverity(undefined)).toBe(false);
|
||||
});
|
||||
|
||||
it("formats a batch with the advisor prefix and severity-tagged bullets", () => {
|
||||
it("wraps each note in an advisory tag with severity as an attribute and escapes the body", () => {
|
||||
const content = formatAdvisorBatchContent([
|
||||
{ note: "first note" },
|
||||
{ note: "second note", severity: "blocker" },
|
||||
{ note: "second <note> & more", severity: "blocker" },
|
||||
]);
|
||||
const lines = content.split("\n");
|
||||
expect(lines[0]).toContain("senior reviewer");
|
||||
expect(lines[1]).toBe("- first note");
|
||||
expect(lines[2]).toBe("- [blocker] second note");
|
||||
// No-severity note: bare advisory tag (no severity attribute).
|
||||
expect(content).toMatch(/<advisory guidance="[^"]*">\nfirst note\n<\/advisory>/);
|
||||
// Severity rides an attribute, not an inline `[blocker]` tag or a bullet.
|
||||
expect(content).toMatch(/<advisory severity="blocker" guidance="[^"]*">/);
|
||||
expect(content).not.toContain("[blocker]");
|
||||
expect(content).not.toContain("- first note");
|
||||
// XML-significant characters in the body are escaped so they can't break the tag.
|
||||
expect(content).toContain("second <note> & more");
|
||||
// Exactly one severity attribute (only the blocker note carries one).
|
||||
expect(content.split('severity="').length - 1).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -279,6 +285,56 @@ describe("advisor", () => {
|
||||
expect(promptInputs[0]).not.toContain("note");
|
||||
});
|
||||
|
||||
it("renders the watched delta with a heading, watched-role labels, and no inner ## headings", () => {
|
||||
const promptInputs: string[] = [];
|
||||
const agent = makeAgent(promptInputs);
|
||||
const messages: AgentMessage[] = [
|
||||
{ role: "user", content: "do the thing", timestamp: 1 } as AgentMessage,
|
||||
{
|
||||
role: "assistant",
|
||||
content: [{ type: "toolCall", id: "a", name: "read", arguments: { path: "x.ts" } }],
|
||||
timestamp: 2,
|
||||
} as unknown as AgentMessage,
|
||||
{
|
||||
role: "toolResult",
|
||||
toolCallId: "a",
|
||||
toolName: "read",
|
||||
content: [{ type: "text", text: "ok" }],
|
||||
isError: false,
|
||||
timestamp: 3,
|
||||
} as AgentMessage,
|
||||
{
|
||||
role: "assistant",
|
||||
content: [{ type: "toolCall", id: "b", name: "search", arguments: { pattern: "y" } }],
|
||||
timestamp: 4,
|
||||
} as unknown as AgentMessage,
|
||||
{
|
||||
role: "toolResult",
|
||||
toolCallId: "b",
|
||||
toolName: "search",
|
||||
content: [{ type: "text", text: "ok" }],
|
||||
isError: false,
|
||||
timestamp: 5,
|
||||
} as AgentMessage,
|
||||
];
|
||||
const host: AdvisorRuntimeHost = {
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
};
|
||||
const runtime = new AdvisorRuntime(agent, host);
|
||||
runtime.onTurnEnd();
|
||||
expect(promptInputs).toHaveLength(1);
|
||||
const prompt = promptInputs[0];
|
||||
expect(prompt).toContain("### Session update");
|
||||
expect(prompt).toContain("**user**:");
|
||||
expect(prompt).toContain("**agent**:");
|
||||
// Inner role headings would collide with the advisor's own turns in the dump.
|
||||
expect(prompt).not.toContain("## assistant");
|
||||
expect(prompt).not.toContain("## user");
|
||||
// Consecutive assistant tool-call messages collapse under a single label.
|
||||
expect(prompt.split("**agent**:").length - 1).toBe(1);
|
||||
});
|
||||
|
||||
it("handles compaction shrink without prompting", () => {
|
||||
const promptInputs: string[] = [];
|
||||
const agent = makeAgent(promptInputs);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core";
|
||||
import { escapeXmlText } from "@oh-my-pi/pi-utils";
|
||||
import { z } from "zod/v4";
|
||||
import adviseDescription from "../prompts/advisor/advise-tool.md" with { type: "text" };
|
||||
|
||||
@@ -33,15 +34,26 @@ export interface AdvisorMessageDetails {
|
||||
}
|
||||
|
||||
/**
|
||||
* Prose framing prepended to every batched advisor message. Kept here so the
|
||||
* non-interrupting YieldQueue dispatcher and the interrupting steer path build
|
||||
* byte-identical content.
|
||||
* Behavioral framing for the watched agent — advice, not orders. Carried as a
|
||||
* tag attribute (rather than a prose header) so the rendered agent-facing output
|
||||
* stays a clean `<advisory>` block. The primary agent's system prompt never
|
||||
* mentions advisories, so this is its only cue for how to treat them.
|
||||
*/
|
||||
const ADVISOR_BATCH_PREFIX = "Advisor (a senior reviewer watching your work — weigh it, don't blindly obey):";
|
||||
const ADVISOR_GUIDANCE = "weigh, don't blindly obey";
|
||||
|
||||
/** Render one advisor card body from a batch of notes (prefix + one bullet per note). */
|
||||
/**
|
||||
* Render a batch of advisor notes as the agent-facing message body: one
|
||||
* `<advisory>` element per note, severity as an attribute. Shared by the
|
||||
* non-interrupting YieldQueue dispatcher and the interrupting steer path so both
|
||||
* build byte-identical content.
|
||||
*/
|
||||
export function formatAdvisorBatchContent(notes: readonly AdvisorNote[]): string {
|
||||
return `${ADVISOR_BATCH_PREFIX}\n${notes.map(n => `- ${n.severity ? `[${n.severity}] ` : ""}${n.note}`).join("\n")}`;
|
||||
return notes
|
||||
.map(n => {
|
||||
const severity = n.severity ? ` severity="${n.severity}"` : "";
|
||||
return `<advisory${severity} guidance="${ADVISOR_GUIDANCE}">\n${escapeXmlText(n.note)}\n</advisory>`;
|
||||
})
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -157,8 +157,13 @@ export class AdvisorRuntime {
|
||||
.filter(m => !(m.role === "custom" && (m as { customType?: string }).customType === "advisor"));
|
||||
this.#lastCount = all.length;
|
||||
if (delta.length === 0) return null;
|
||||
const md = formatSessionHistoryMarkdown(delta, { includeThinking: true, includeToolIntent: true });
|
||||
return md.trim() ? md : null;
|
||||
const md = formatSessionHistoryMarkdown(delta, {
|
||||
includeThinking: true,
|
||||
includeToolIntent: true,
|
||||
watchedRoles: true,
|
||||
});
|
||||
if (!md.trim()) return null;
|
||||
return `### Session update\n\n${md}`;
|
||||
}
|
||||
|
||||
#notifyWaiters(): void {
|
||||
@@ -182,7 +187,9 @@ export class AdvisorRuntime {
|
||||
try {
|
||||
while (!this.disposed && this.#pending.length) {
|
||||
const popped = this.#pending.splice(0);
|
||||
const candidateBatch = popped.map(b => b.text).join("\n\n---\n\n");
|
||||
// Each delta already opens with a `### Session update` heading, so
|
||||
// join with a blank line rather than a `---` rule.
|
||||
const candidateBatch = popped.map(b => b.text).join("\n\n");
|
||||
const turnsCovered = popped.reduce((sum, b) => sum + b.turns, 0);
|
||||
const incomingTokens = estimateTokens({
|
||||
role: "user",
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { prompt, Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import { escapeXmlText, prompt, Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import goalBudgetLimitPrompt from "../prompts/goals/goal-budget-limit.md" with { type: "text" };
|
||||
import goalContinuationPrompt from "../prompts/goals/goal-continuation.md" with { type: "text" };
|
||||
import goalModeActivePrompt from "../prompts/goals/goal-mode-active.md" with { type: "text" };
|
||||
@@ -58,28 +58,6 @@ export function remainingTokens(goal: Goal | null | undefined): number | null {
|
||||
return Math.max(0, goal.tokenBudget - goal.tokensUsed);
|
||||
}
|
||||
|
||||
export function escapeXmlText(input: string): string {
|
||||
let firstEscapable = -1;
|
||||
for (let index = 0; index < input.length; index++) {
|
||||
const char = input.charCodeAt(index);
|
||||
if (char === 38 || char === 60 || char === 62) {
|
||||
firstEscapable = index;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (firstEscapable === -1) return input;
|
||||
|
||||
let output = input.slice(0, firstEscapable);
|
||||
for (let index = firstEscapable; index < input.length; index++) {
|
||||
const char = input[index];
|
||||
if (char === "&") output += "&";
|
||||
else if (char === "<") output += "<";
|
||||
else if (char === ">") output += ">";
|
||||
else output += char;
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
export function renderTrustedObjective(objective: string): string {
|
||||
return `<objective>\n${escapeXmlText(objective)}\n</objective>`;
|
||||
}
|
||||
|
||||
@@ -26,6 +26,8 @@ export interface HistoryFormatOptions {
|
||||
includeThinking?: boolean;
|
||||
/** Render tool intent comment before tool call lines. */
|
||||
includeToolIntent?: boolean;
|
||||
/** Render watched-session roles as inline `**agent**:` / `**user**:` labels (collapsing consecutive same-role messages) instead of `## ` headings, so a primary transcript embedded inside an advisor turn stays visually distinct. */
|
||||
watchedRoles?: boolean;
|
||||
}
|
||||
|
||||
/** Max length of the primary-arg summary inside `→ tool(...)` lines. */
|
||||
@@ -125,7 +127,7 @@ function toolCallLine(
|
||||
const intent = includeToolIntent ? args?.[INTENT_FIELD] : undefined;
|
||||
if (typeof intent === "string" && intent.trim()) {
|
||||
const formattedIntent = oneLine(intent, 80);
|
||||
return `# ${formattedIntent}\n${base}`;
|
||||
return `// ${formattedIntent}\n${base}`;
|
||||
}
|
||||
return base;
|
||||
}
|
||||
@@ -191,6 +193,11 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History
|
||||
}
|
||||
}
|
||||
const consumed = new Set<string>();
|
||||
// In watched mode, consecutive same-role messages collapse under one label
|
||||
// (the watched agent emits one assistant message per tool call, so otherwise
|
||||
// every call repeats `**agent**:`). Cleared whenever a
|
||||
// non-role-labeled line is emitted so the next turn re-labels.
|
||||
let lastWatchedLabel: string | undefined;
|
||||
|
||||
for (const msg of typed) {
|
||||
switch (msg.role) {
|
||||
@@ -198,7 +205,17 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History
|
||||
case "developer": {
|
||||
const text = contentToText(msg.content);
|
||||
if (!text.trim()) break;
|
||||
lines.push(`## ${msg.role}`, "", text, "");
|
||||
if (opts?.watchedRoles) {
|
||||
const label = `**${msg.role}**:`;
|
||||
if (lastWatchedLabel === label) {
|
||||
lines.push(text, "");
|
||||
} else {
|
||||
lines.push(label, text, "");
|
||||
lastWatchedLabel = label;
|
||||
}
|
||||
} else {
|
||||
lines.push(`## ${msg.role}`, "", text, "");
|
||||
}
|
||||
break;
|
||||
}
|
||||
case "assistant": {
|
||||
@@ -217,45 +234,62 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History
|
||||
// redactedThinking elided entirely (no readable text)
|
||||
}
|
||||
if (body.length === 0) break;
|
||||
lines.push("## assistant", "", ...body, "");
|
||||
if (opts?.watchedRoles) {
|
||||
const label = "**agent**:";
|
||||
if (lastWatchedLabel === label) {
|
||||
lines.push(...body, "");
|
||||
} else {
|
||||
lines.push(label, ...body, "");
|
||||
lastWatchedLabel = label;
|
||||
}
|
||||
} else {
|
||||
lines.push("## assistant", "", ...body, "");
|
||||
}
|
||||
break;
|
||||
}
|
||||
case "toolResult": {
|
||||
// Normally consumed by its toolCall; orphans (e.g. truncated history) get their own line.
|
||||
if (consumed.has(msg.toolCallId)) break;
|
||||
lines.push(toolCallLine(msg.toolName, undefined, msg, opts?.includeToolIntent), "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
case "bashExecution": {
|
||||
const bashMsg = msg as BashExecutionMessage;
|
||||
if (bashMsg.excludeFromContext) break;
|
||||
lines.push(executionLine("bash", bashMsg.command, bashMsg), "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
case "pythonExecution": {
|
||||
const pythonMsg = msg as PythonExecutionMessage;
|
||||
if (pythonMsg.excludeFromContext) break;
|
||||
lines.push(executionLine("python", pythonMsg.code, pythonMsg), "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
case "custom":
|
||||
case "hookMessage": {
|
||||
lines.push(customOneLiner(msg as CustomMessage | HookMessage), "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
case "branchSummary": {
|
||||
const branchMsg = msg as BranchSummaryMessage;
|
||||
lines.push(`[branch] from ${branchMsg.fromId}: ${oneLine(branchMsg.summary)}`, "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
case "compactionSummary": {
|
||||
const compactMsg = msg as CompactionSummaryMessage;
|
||||
lines.push(`[compaction] ${oneLine(compactMsg.summary)}`, "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
case "fileMention": {
|
||||
const fileMsg = msg as FileMentionMessage;
|
||||
lines.push(`[file-mention] ${oneLine(fileMsg.files.map(f => f.path).join(", "))}`, "");
|
||||
lastWatchedLabel = undefined;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import {
|
||||
escapeXmlText,
|
||||
GoalRuntime,
|
||||
type GoalRuntimeHost,
|
||||
goalTokenDelta,
|
||||
@@ -8,6 +7,7 @@ import {
|
||||
renderTrustedObjective,
|
||||
} from "@oh-my-pi/pi-coding-agent/goals/runtime";
|
||||
import type { Goal, GoalModeState, GoalRuntimeEvent, GoalTokenUsage } from "@oh-my-pi/pi-coding-agent/goals/state";
|
||||
import { escapeXmlText } from "@oh-my-pi/pi-utils";
|
||||
|
||||
function createUsage(overrides: Partial<GoalTokenUsage> = {}): GoalTokenUsage {
|
||||
return {
|
||||
|
||||
@@ -92,6 +92,14 @@ describe("formatSessionHistoryMarkdown", () => {
|
||||
expect(output.startsWith("# Spawnling (idle)\n")).toBe(true);
|
||||
});
|
||||
|
||||
it("renders watched roles using bold text rather than level-2 headers when watchedRoles is true", () => {
|
||||
const output = formatSessionHistoryMarkdown(buildMessages(), { watchedRoles: true });
|
||||
expect(output).toContain("**user**:");
|
||||
expect(output).toContain("**agent**:");
|
||||
expect(output).not.toContain("## user");
|
||||
expect(output).not.toContain("## assistant");
|
||||
});
|
||||
|
||||
it("renders an orphan toolResult (truncated history) as its own line", () => {
|
||||
const output = formatSessionHistoryMarkdown([
|
||||
{
|
||||
@@ -145,14 +153,14 @@ describe("formatSessionHistoryMarkdown", () => {
|
||||
];
|
||||
|
||||
const outputWithIntent = formatSessionHistoryMarkdown(messages, { includeToolIntent: true });
|
||||
expect(outputWithIntent).toContain("# reading config file\n→ read(src/config.ts) ⇒ ok · 1 line");
|
||||
expect(outputWithIntent).toContain("// reading config file\n→ read(src/config.ts) ⇒ ok · 1 line");
|
||||
// The long intent should be flattened to one line and truncated to 80 characters (including ellipsis).
|
||||
expect(outputWithIntent).toContain(
|
||||
"# reading config file with a very very long and descriptive intent that will exce…\n→ read(src/config.ts) ⇒ ok · 1 line",
|
||||
"// reading config file with a very very long and descriptive intent that will exce…\n→ read(src/config.ts) ⇒ ok · 1 line",
|
||||
);
|
||||
|
||||
const outputWithoutIntent = formatSessionHistoryMarkdown(messages);
|
||||
expect(outputWithoutIntent).not.toContain("# reading config file");
|
||||
expect(outputWithoutIntent).not.toContain("// reading config file");
|
||||
expect(outputWithoutIntent).toContain("→ read(src/config.ts) ⇒ ok · 1 line");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -518,10 +518,16 @@ async function main(): Promise<void> {
|
||||
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
|
||||
);
|
||||
console.log(
|
||||
` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
|
||||
` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (best total): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`,
|
||||
` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`,
|
||||
);
|
||||
console.log(
|
||||
` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total}`,
|
||||
);
|
||||
if (result.summary.ghostRuns > 0) {
|
||||
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
|
||||
@@ -563,6 +569,7 @@ class LiveProgress {
|
||||
#inputTokens: number[] = [];
|
||||
#outputTokens: number[] = [];
|
||||
#totalTokens: number[] = [];
|
||||
#oneShotSuccessTokens: number[] = [];
|
||||
#lastLineLength = 0;
|
||||
|
||||
constructor(totalRuns: number, runsPerTask: number) {
|
||||
@@ -586,6 +593,9 @@ class LiveProgress {
|
||||
if (event.result.success) {
|
||||
this.#success += 1;
|
||||
}
|
||||
if (event.result.success && event.runIndex === 0) {
|
||||
this.#oneShotSuccessTokens.push(event.result.tokens.total);
|
||||
}
|
||||
this.#totalInput += event.result.tokens.input;
|
||||
this.#totalOutput += event.result.tokens.output;
|
||||
this.#inputTokens.push(event.result.tokens.input);
|
||||
@@ -689,6 +699,7 @@ class LiveProgress {
|
||||
console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
|
||||
console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
|
||||
console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
|
||||
console.log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`);
|
||||
console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
|
||||
}
|
||||
|
||||
|
||||
@@ -172,7 +172,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
`| **Tool Input Chars** | ${formatNumber(summary.totalToolCalls.totalInputChars)} | ${formatNumber(Math.round(summary.avgToolCallsPerTask.totalInputChars))} |`,
|
||||
);
|
||||
lines.push("");
|
||||
lines.push("### Tokens & Time");
|
||||
lines.push("### Tokens & Time (Overall)");
|
||||
lines.push("");
|
||||
lines.push("| Metric | Total (best) | Avg/Task | Median | P1 | P99 |");
|
||||
lines.push("|--------|--------------|----------|--------|----|----|");
|
||||
@@ -190,6 +190,20 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
);
|
||||
lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** | — | — | — |`);
|
||||
lines.push("");
|
||||
lines.push("### Tokens & Time (One-shot Successes)");
|
||||
lines.push("");
|
||||
lines.push("| Metric | Total | Avg/Task | Median | P1 | P99 |");
|
||||
lines.push("|--------|-------|----------|--------|----|----|");
|
||||
lines.push(
|
||||
`| Input Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.input)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.input)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.input)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.input)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.input)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Output Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.output)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.output)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Total Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.total)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.total)} |`,
|
||||
);
|
||||
lines.push("");
|
||||
|
||||
if (summary.hashlineEditSubtypes) {
|
||||
const order = ["set", "set_range", "insert"] as const;
|
||||
|
||||
@@ -875,6 +875,18 @@ export interface BenchmarkSummary {
|
||||
flakyTasks: number;
|
||||
/** Tasks where every executed non-ghost run succeeded. */
|
||||
consistentlyPassingTasks: number;
|
||||
/** Tasks whose first run succeeded. */
|
||||
successfulOneShotTasks: number;
|
||||
/** Tokens summed over the first run of each successfully one-shot task. */
|
||||
totalOneShotSuccessTokens: TokenStats;
|
||||
/** Average tokens per successfully one-shot task. */
|
||||
avgOneShotSuccessTokensPerTask: TokenStats;
|
||||
/** Median tokens across successfully one-shot tasks. */
|
||||
medianOneShotSuccessTokensPerTask: TokenStats;
|
||||
/** 1st-percentile tokens across successfully one-shot tasks. */
|
||||
p1OneShotSuccessTokensPerTask: TokenStats;
|
||||
/** 99th-percentile tokens across successfully one-shot tasks. */
|
||||
p99OneShotSuccessTokensPerTask: TokenStats;
|
||||
/** Tokens summed over the best run of each task. */
|
||||
totalTokens: TokenStats;
|
||||
/** Average tokens per task (sum of best runs / number of tasks). */
|
||||
@@ -1943,8 +1955,31 @@ export function buildBenchmarkResult(params: {
|
||||
? bestWithMutationIntent.filter(r => r.mutationIntentMatched).length / bestWithMutationIntent.length
|
||||
: undefined;
|
||||
|
||||
const oneShotSuccessRuns = taskResults
|
||||
.map(t => t.runs.find(r => r.runIndex === 0))
|
||||
.filter((r): r is TaskRunResult => Boolean(r?.success));
|
||||
const successfulOneShotTasks = oneShotSuccessRuns.length;
|
||||
const oneShotDenom = successfulOneShotTasks || 1;
|
||||
|
||||
const totalOneShotSuccessTokens: TokenStats = {
|
||||
input: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.input, 0),
|
||||
output: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.output, 0),
|
||||
total: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.total, 0),
|
||||
};
|
||||
const oneShotTokenDistribution = summarizeTokenDistribution(oneShotSuccessRuns);
|
||||
|
||||
const taskDenom = tasksWithBestRun || 1;
|
||||
const summary: BenchmarkSummary = {
|
||||
successfulOneShotTasks,
|
||||
totalOneShotSuccessTokens,
|
||||
avgOneShotSuccessTokensPerTask: {
|
||||
input: Math.round(totalOneShotSuccessTokens.input / oneShotDenom),
|
||||
output: Math.round(totalOneShotSuccessTokens.output / oneShotDenom),
|
||||
total: Math.round(totalOneShotSuccessTokens.total / oneShotDenom),
|
||||
},
|
||||
medianOneShotSuccessTokensPerTask: oneShotTokenDistribution.median,
|
||||
p1OneShotSuccessTokensPerTask: oneShotTokenDistribution.p1,
|
||||
p99OneShotSuccessTokensPerTask: oneShotTokenDistribution.p99,
|
||||
totalTasks,
|
||||
totalRuns,
|
||||
successfulRuns,
|
||||
|
||||
@@ -310,6 +310,56 @@ describe("buildBenchmarkResult", () => {
|
||||
expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, total: 114 });
|
||||
expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, total: 546 });
|
||||
});
|
||||
|
||||
it("separates token stats for successfully one-shot tasks vs overall", () => {
|
||||
// Task 1: Succeeded on run 0 (one-shot success). Tokens: 100
|
||||
// Task 2: Failed on run 0 (150 tokens), succeeded on run 1 (best run, 50 tokens).
|
||||
// Task 3: Failed on run 0 (200 tokens).
|
||||
const tasks = [createTask("t1"), createTask("t2"), createTask("t3")];
|
||||
const resultsByTask = new Map([
|
||||
["t1", [createRun(0, true, { tokens: { input: 80, output: 20, total: 100 } })]],
|
||||
[
|
||||
"t2",
|
||||
[
|
||||
createRun(0, false, { tokens: { input: 120, output: 30, total: 150 } }),
|
||||
createRun(1, true, { tokens: { input: 40, output: 10, total: 50 } }),
|
||||
],
|
||||
],
|
||||
["t3", [createRun(0, false, { tokens: { input: 160, output: 40, total: 200 } })]],
|
||||
]);
|
||||
|
||||
const result = buildBenchmarkResult({
|
||||
tasks,
|
||||
config: {
|
||||
provider: "anthropic",
|
||||
model: "claude",
|
||||
runsPerTask: 2,
|
||||
timeout: 1000,
|
||||
taskConcurrency: 1,
|
||||
},
|
||||
resultsByTask,
|
||||
startTime: "2026-04-28T00:00:00.000Z",
|
||||
endTime: "2026-04-28T00:00:01.000Z",
|
||||
});
|
||||
|
||||
const { summary } = result;
|
||||
// Overall uses best runs:
|
||||
// t1 best run: run 0 (100 tokens, success)
|
||||
// t2 best run: run 1 (50 tokens, success)
|
||||
// t3 best run: run 0 (200 tokens, fail)
|
||||
// Total overall tokens: 100 + 50 + 200 = 350
|
||||
expect(summary.totalTokens.total).toBe(350);
|
||||
expect(summary.avgTokensPerTask.total).toBe(Math.round(350 / 3)); // 117
|
||||
|
||||
// Successfully one-shot tasks (run 0 succeeded):
|
||||
// t1 succeeded on run 0 (100 tokens)
|
||||
// t2 failed on run 0
|
||||
// t3 failed on run 0
|
||||
// Only t1 counts.
|
||||
expect(summary.successfulOneShotTasks).toBe(1);
|
||||
expect(summary.totalOneShotSuccessTokens.total).toBe(100);
|
||||
expect(summary.avgOneShotSuccessTokensPerTask.total).toBe(100);
|
||||
});
|
||||
});
|
||||
|
||||
describe("writeConversationDump", () => {
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Added
|
||||
|
||||
- Added `escapeXmlText` utility to escape XML-significant characters `&`, `<`, and `>` in element body text
|
||||
|
||||
## [15.13.3] - 2026-06-15
|
||||
|
||||
@@ -143,4 +146,4 @@
|
||||
|
||||
### Added
|
||||
|
||||
- Added an XDG-aware tiny-title model cache directory helper for coding-agent local title models.
|
||||
- Added an XDG-aware tiny-title model cache directory helper for coding-agent local title models.
|
||||
@@ -36,3 +36,31 @@ function sanitizeWellFormedText(text: string): string {
|
||||
CONTROL_RE.lastIndex = 0;
|
||||
return stripped.replace(CONTROL_RE, "");
|
||||
}
|
||||
|
||||
/**
|
||||
* Escape the three XML-significant characters (`&`, `<`, `>`) in text destined
|
||||
* for an XML/markup element body. Allocation-conscious: returns the input
|
||||
* unchanged (same reference) when nothing needs escaping. Quotes are left as-is
|
||||
* — use it for element text, not attribute values.
|
||||
*/
|
||||
export function escapeXmlText(input: string): string {
|
||||
let firstEscapable = -1;
|
||||
for (let index = 0; index < input.length; index++) {
|
||||
const char = input.charCodeAt(index);
|
||||
if (char === 38 || char === 60 || char === 62) {
|
||||
firstEscapable = index;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (firstEscapable === -1) return input;
|
||||
|
||||
let output = input.slice(0, firstEscapable);
|
||||
for (let index = firstEscapable; index < input.length; index++) {
|
||||
const char = input[index];
|
||||
if (char === "&") output += "&";
|
||||
else if (char === "<") output += "<";
|
||||
else if (char === ">") output += ">";
|
||||
else output += char;
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user