diff --git a/.omp/skills/tool-prompt-optimization/SKILL.md b/.omp/skills/tool-prompt-optimization/SKILL.md index 2e09ed8ab..b45244ac7 100644 --- a/.omp/skills/tool-prompt-optimization/SKILL.md +++ b/.omp/skills/tool-prompt-optimization/SKILL.md @@ -20,9 +20,22 @@ bun .omp/skills/tool-prompt-optimization/scripts/probe.ts \ - `--schema` and `--template` are the only required inputs (file path or inline value). - No `--model` → 3-model panel (`fireworks/kimi-k2.7-code`, `anthropic/claude-opus-4-8`, `openai/gpt-5.5`) × `--samples` (default 3). Needs `FIREWORKS_API_KEY` / `ANTHROPIC_API_KEY` / `OPENAI_API_KEY`. -- `--model p/id,p/id` overrides the panel; `--samples N`, `--temp`, `--max-tokens`, `--json` tune it. +- `--model p/id,p/id` overrides the panel; `--samples N`, `--max-tokens`, `--json` tune it. - Programmatic: `import { probe } from "./scripts/probe.ts"` → `{ prompt, results: [{ model, samples: [{ text, stopReason, usage, error }] }] }`. +### Builtin shortcut (preferred for this repo's tools) + +Skip building the two inputs by hand — `scripts/probe-builtin.ts` instantiates the live tool, pulls the EXACT wire schema (`toolWireSchema`) and rendered prompt (`tool.description`), and derives the outline for you: + +```bash +bun .omp/skills/tool-prompt-optimization/scripts/probe-builtin.ts --tool [--no-summary] [--show] +``` + +- `--show` prints the resolved schema + derived outline + real prompt and exits (no API calls) — use it to eyeball inputs before spending tokens. +- `--no-summary` runs the ablation (blank the summary line) directly. +- `--samples` / `--model` / `--max-tokens` / `--json` forward to the panel; output ends with the REAL prompt so you can diff in place. +- It bypasses the settings allowlist via the factory map, so gated tools (`irc`, `github`, …) resolve. If a tool refuses to construct (an availability gate like a missing `gh` CLI), fall back to the manual inputs below. + ## Build the two inputs **Schema** — use the *wire* schema the model actually sees, not a hand-sketch. For this repo's arktype tool schemas: diff --git a/.omp/skills/tool-prompt-optimization/scripts/probe-builtin.ts b/.omp/skills/tool-prompt-optimization/scripts/probe-builtin.ts new file mode 100755 index 000000000..189408b55 --- /dev/null +++ b/.omp/skills/tool-prompt-optimization/scripts/probe-builtin.ts @@ -0,0 +1,162 @@ +#!/usr/bin/env bun +/** + * Builtin shortcut for the prompt-inference probe. + * + * Instead of hand-writing a tool's JSON schema + outline, point this at a live + * builtin tool name. It resolves the tool (direct constructor for availability-gated + * `github`/`irc`, else the BUILTIN_TOOLS/HIDDEN_TOOLS factory map), pulls the EXACT wire + * schema the model sees (`toolWireSchema`) and the rendered prompt (`tool.description`), + * derives an outline by blanking section bodies, then runs the same `probe()` panel. + * + * Usage: + * bun probe-builtin.ts --tool [--no-summary] [--show] + * --tool builtin tool name (e.g. irc, github, read). Required. + * --no-summary ablation: blank the one-line summary too (isolate schema-alone). + * --show print resolved schema + outline + real prompt and exit (no API calls). + * --samples / --model / --max-tokens / --json forwarded to probe(). + * + * The heavy coding-agent import lives here; probe.ts stays pi-ai-only. + */ +import { parseArgs } from "node:util"; +import { toolWireSchema } from "@oh-my-pi/pi-ai"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { BUILTIN_TOOLS, GithubTool, HIDDEN_TOOLS, IrcTool, type Tool, type ToolFactory, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { probe } from "./probe.ts"; + +const OPEN_TAG = /^<[a-z_][\w-]*>$/i; +const CLOSE_TAG = /^<\/[a-z_][\w-]*>$/i; +const MD_HEADER = /^#{1,6}\s/; + +/** Keep the summary + every section header/tag; collapse each body run to a single `...`. */ +function deriveOutline(description: string, dropSummary: boolean): string { + const lines = description.split("\n"); + let i = 0; + const summary: string[] = []; + const isHeader = (l: string): boolean => { + const t = l.trim(); + return OPEN_TAG.test(t) || CLOSE_TAG.test(t) || MD_HEADER.test(l); + }; + while (i < lines.length && !isHeader(lines[i])) { + summary.push(lines[i]); + i++; + } + const parts: string[] = [dropSummary ? "" : summary.join("\n").trim()]; + let pendingBody = false; + for (; i < lines.length; i++) { + const line = lines[i]; + const t = line.trim(); + if (OPEN_TAG.test(t) || MD_HEADER.test(line)) { + parts.push("", line); + pendingBody = false; + } else if (CLOSE_TAG.test(t)) { + parts.push(line); + pendingBody = false; + } else if (t !== "") { + if (!pendingBody) { + parts.push("..."); + pendingBody = true; + } + } + } + return parts.join("\n").trim(); +} + +async function resolveTool(name: string): Promise { + await Settings.init({ inMemory: true, cwd: process.cwd() }); + const settings = Settings.isolated({}); + // Enriched stub: we only read `.parameters` / `.description`, never `execute`, so the + // availability-gated factories (irc needs a registry + agent id; github needs `gh`) can + // still construct. Factory map bypasses the settings allowlist (`isToolAllowed`). + const session = { + cwd: process.cwd(), + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings, + taskDepth: 0, + getAgentId: () => "Probe", + agentRegistry: {}, + } as ToolSession; + // `github`/`irc` map to `*.createIf`, which gates on external availability (gh CLI) or a + // live agent registry. We only read `.parameters`/`.description`, so direct-construct those + // two — keeps the probe gh-independent. Everything else goes through the factory map. + const direct: Record Tool> = { + github: s => new GithubTool(s), + irc: s => new IrcTool(s), + }; + const key = name.toLowerCase(); + const directCtor = direct[key]; + if (directCtor) return directCtor(session); + const factories: Record = { ...BUILTIN_TOOLS, ...HIDDEN_TOOLS }; + const factory = factories[key]; + if (!factory) { + const names = Object.keys(factories).sort().join(", "); + throw new Error(`unknown builtin tool "${name}". available: ${names}`); + } + const tool = await factory(session); + if (!tool) { + throw new Error(`tool "${name}" did not construct here — blocked by an availability gate (e.g. ssh, or a memory backend that isn't configured). Fall back to the manual --schema/--template path.`); + } + return tool; +} + +async function main(): Promise { + const { values } = parseArgs({ + args: Bun.argv.slice(2), + options: { + tool: { type: "string" }, + "no-summary": { type: "boolean" }, + show: { type: "boolean" }, + samples: { type: "string" }, + model: { type: "string" }, + "max-tokens": { type: "string" }, + json: { type: "boolean" }, + }, + allowPositionals: false, + }); + + if (!values.tool) { + console.error("usage: bun probe-builtin.ts --tool [--no-summary] [--show] [--samples N] [--model p/id,...] [--max-tokens N] [--json]"); + process.exit(2); + } + + const tool = await resolveTool(values.tool); + const schema = toolWireSchema(tool); + const realPrompt = tool.description ?? ""; + const outline = deriveOutline(realPrompt, Boolean(values["no-summary"])); + + if (values.show) { + console.log(`# tool: ${tool.name}\n\n## wire schema\n\`\`\`json\n${JSON.stringify(schema, null, 2)}\n\`\`\`\n`); + console.log(`## derived outline\n${outline}\n`); + console.log(`## real prompt (${Buffer.byteLength(realPrompt)} bytes)\n${realPrompt}`); + return; + } + + const run = await probe({ + schema, + template: outline, + name: tool.name, + samples: values.samples ? Number(values.samples) : undefined, + models: values.model ? values.model.split(",").map(s => s.trim()).filter(Boolean) : undefined, + maxTokens: values["max-tokens"] ? Number(values["max-tokens"]) : undefined, + }); + + if (values.json) { + console.log(JSON.stringify({ ...run, realPrompt }, null, 2)); + return; + } + + for (const result of run.results) { + console.log(`\n############ ${result.model} ############`); + result.samples.forEach((s, i) => { + const tag = s.error ? `ERROR: ${s.error}` : s.stopReason; + console.log(`\n----- sample ${i + 1}/${result.samples.length} [${tag}] -----`); + console.log(s.error ? "" : s.text); + }); + } + console.log(`\n############ REAL PROMPT (${tool.name}) ############\n${realPrompt}`); +} + +if (import.meta.main) { + await main(); +}