feat(skills/tool-prompt-optimization): added builtin probing utility

- Added `probe-builtin.ts` to allow testing tools directly by name instead of manual schema and template input.
- Implemented automatic wire schema resolution and prompt outlining for builtin tools.
- Updated documentation with usage instructions for the new shortcut script.
This commit is contained in:
can1357
2026-06-19 15:59:52 +02:00
parent a98230addc
commit a1a07fa9e1
2 changed files with 176 additions and 1 deletions
+14 -1
View File
@@ -20,9 +20,22 @@ bun .omp/skills/tool-prompt-optimization/scripts/probe.ts \
- `--schema` and `--template` are the only required inputs (file path or inline value).
- No `--model` → 3-model panel (`fireworks/kimi-k2.7-code`, `anthropic/claude-opus-4-8`, `openai/gpt-5.5`) × `--samples` (default 3). Needs `FIREWORKS_API_KEY` / `ANTHROPIC_API_KEY` / `OPENAI_API_KEY`.
- `--model p/id,p/id` overrides the panel; `--samples N`, `--temp`, `--max-tokens`, `--json` tune it.
- `--model p/id,p/id` overrides the panel; `--samples N`, `--max-tokens`, `--json` tune it.
- Programmatic: `import { probe } from "./scripts/probe.ts"` → `{ prompt, results: [{ model, samples: [{ text, stopReason, usage, error }] }] }`.
### Builtin shortcut (preferred for this repo's tools)
Skip building the two inputs by hand — `scripts/probe-builtin.ts` instantiates the live tool, pulls the EXACT wire schema (`toolWireSchema`) and rendered prompt (`tool.description`), and derives the outline for you:
```bash
bun .omp/skills/tool-prompt-optimization/scripts/probe-builtin.ts --tool <name> [--no-summary] [--show]
```
- `--show` prints the resolved schema + derived outline + real prompt and exits (no API calls) — use it to eyeball inputs before spending tokens.
- `--no-summary` runs the ablation (blank the summary line) directly.
- `--samples` / `--model` / `--max-tokens` / `--json` forward to the panel; output ends with the REAL prompt so you can diff in place.
- It bypasses the settings allowlist via the factory map, so gated tools (`irc`, `github`, …) resolve. If a tool refuses to construct (an availability gate like a missing `gh` CLI), fall back to the manual inputs below.
## Build the two inputs
**Schema** — use the *wire* schema the model actually sees, not a hand-sketch. For this repo's arktype tool schemas:
@@ -0,0 +1,162 @@
#!/usr/bin/env bun
/**
* Builtin shortcut for the prompt-inference probe.
*
* Instead of hand-writing a tool's JSON schema + outline, point this at a live
* builtin tool name. It resolves the tool (direct constructor for availability-gated
* `github`/`irc`, else the BUILTIN_TOOLS/HIDDEN_TOOLS factory map), pulls the EXACT wire
* schema the model sees (`toolWireSchema`) and the rendered prompt (`tool.description`),
* derives an outline by blanking section bodies, then runs the same `probe()` panel.
*
* Usage:
* bun probe-builtin.ts --tool <name> [--no-summary] [--show]
* --tool <name> builtin tool name (e.g. irc, github, read). Required.
* --no-summary ablation: blank the one-line summary too (isolate schema-alone).
* --show print resolved schema + outline + real prompt and exit (no API calls).
* --samples / --model / --max-tokens / --json forwarded to probe().
*
* The heavy coding-agent import lives here; probe.ts stays pi-ai-only.
*/
import { parseArgs } from "node:util";
import { toolWireSchema } from "@oh-my-pi/pi-ai";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { BUILTIN_TOOLS, GithubTool, HIDDEN_TOOLS, IrcTool, type Tool, type ToolFactory, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { probe } from "./probe.ts";
const OPEN_TAG = /^<[a-z_][\w-]*>$/i;
const CLOSE_TAG = /^<\/[a-z_][\w-]*>$/i;
const MD_HEADER = /^#{1,6}\s/;
/** Keep the summary + every section header/tag; collapse each body run to a single `...`. */
function deriveOutline(description: string, dropSummary: boolean): string {
const lines = description.split("\n");
let i = 0;
const summary: string[] = [];
const isHeader = (l: string): boolean => {
const t = l.trim();
return OPEN_TAG.test(t) || CLOSE_TAG.test(t) || MD_HEADER.test(l);
};
while (i < lines.length && !isHeader(lines[i])) {
summary.push(lines[i]);
i++;
}
const parts: string[] = [dropSummary ? "" : summary.join("\n").trim()];
let pendingBody = false;
for (; i < lines.length; i++) {
const line = lines[i];
const t = line.trim();
if (OPEN_TAG.test(t) || MD_HEADER.test(line)) {
parts.push("", line);
pendingBody = false;
} else if (CLOSE_TAG.test(t)) {
parts.push(line);
pendingBody = false;
} else if (t !== "") {
if (!pendingBody) {
parts.push("...");
pendingBody = true;
}
}
}
return parts.join("\n").trim();
}
async function resolveTool(name: string): Promise<Tool> {
await Settings.init({ inMemory: true, cwd: process.cwd() });
const settings = Settings.isolated({});
// Enriched stub: we only read `.parameters` / `.description`, never `execute`, so the
// availability-gated factories (irc needs a registry + agent id; github needs `gh`) can
// still construct. Factory map bypasses the settings allowlist (`isToolAllowed`).
const session = {
cwd: process.cwd(),
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings,
taskDepth: 0,
getAgentId: () => "Probe",
agentRegistry: {},
} as ToolSession;
// `github`/`irc` map to `*.createIf`, which gates on external availability (gh CLI) or a
// live agent registry. We only read `.parameters`/`.description`, so direct-construct those
// two — keeps the probe gh-independent. Everything else goes through the factory map.
const direct: Record<string, (s: ToolSession) => Tool> = {
github: s => new GithubTool(s),
irc: s => new IrcTool(s),
};
const key = name.toLowerCase();
const directCtor = direct[key];
if (directCtor) return directCtor(session);
const factories: Record<string, ToolFactory> = { ...BUILTIN_TOOLS, ...HIDDEN_TOOLS };
const factory = factories[key];
if (!factory) {
const names = Object.keys(factories).sort().join(", ");
throw new Error(`unknown builtin tool "${name}". available: ${names}`);
}
const tool = await factory(session);
if (!tool) {
throw new Error(`tool "${name}" did not construct here — blocked by an availability gate (e.g. ssh, or a memory backend that isn't configured). Fall back to the manual --schema/--template path.`);
}
return tool;
}
async function main(): Promise<void> {
const { values } = parseArgs({
args: Bun.argv.slice(2),
options: {
tool: { type: "string" },
"no-summary": { type: "boolean" },
show: { type: "boolean" },
samples: { type: "string" },
model: { type: "string" },
"max-tokens": { type: "string" },
json: { type: "boolean" },
},
allowPositionals: false,
});
if (!values.tool) {
console.error("usage: bun probe-builtin.ts --tool <name> [--no-summary] [--show] [--samples N] [--model p/id,...] [--max-tokens N] [--json]");
process.exit(2);
}
const tool = await resolveTool(values.tool);
const schema = toolWireSchema(tool);
const realPrompt = tool.description ?? "";
const outline = deriveOutline(realPrompt, Boolean(values["no-summary"]));
if (values.show) {
console.log(`# tool: ${tool.name}\n\n## wire schema\n\`\`\`json\n${JSON.stringify(schema, null, 2)}\n\`\`\`\n`);
console.log(`## derived outline\n${outline}\n`);
console.log(`## real prompt (${Buffer.byteLength(realPrompt)} bytes)\n${realPrompt}`);
return;
}
const run = await probe({
schema,
template: outline,
name: tool.name,
samples: values.samples ? Number(values.samples) : undefined,
models: values.model ? values.model.split(",").map(s => s.trim()).filter(Boolean) : undefined,
maxTokens: values["max-tokens"] ? Number(values["max-tokens"]) : undefined,
});
if (values.json) {
console.log(JSON.stringify({ ...run, realPrompt }, null, 2));
return;
}
for (const result of run.results) {
console.log(`\n############ ${result.model} ############`);
result.samples.forEach((s, i) => {
const tag = s.error ? `ERROR: ${s.error}` : s.stopReason;
console.log(`\n----- sample ${i + 1}/${result.samples.length} [${tag}] -----`);
console.log(s.error ? "" : s.text);
});
}
console.log(`\n############ REAL PROMPT (${tool.name}) ############\n${realPrompt}`);
}
if (import.meta.main) {
await main();
}