feat(skills/tool-prompt-optimization): added builtin probing utility
- Added `probe-builtin.ts` to allow testing tools directly by name instead of manual schema and template input. - Implemented automatic wire schema resolution and prompt outlining for builtin tools. - Updated documentation with usage instructions for the new shortcut script.
This commit is contained in:
@@ -20,9 +20,22 @@ bun .omp/skills/tool-prompt-optimization/scripts/probe.ts \
|
||||
|
||||
- `--schema` and `--template` are the only required inputs (file path or inline value).
|
||||
- No `--model` → 3-model panel (`fireworks/kimi-k2.7-code`, `anthropic/claude-opus-4-8`, `openai/gpt-5.5`) × `--samples` (default 3). Needs `FIREWORKS_API_KEY` / `ANTHROPIC_API_KEY` / `OPENAI_API_KEY`.
|
||||
- `--model p/id,p/id` overrides the panel; `--samples N`, `--temp`, `--max-tokens`, `--json` tune it.
|
||||
- `--model p/id,p/id` overrides the panel; `--samples N`, `--max-tokens`, `--json` tune it.
|
||||
- Programmatic: `import { probe } from "./scripts/probe.ts"` → `{ prompt, results: [{ model, samples: [{ text, stopReason, usage, error }] }] }`.
|
||||
|
||||
### Builtin shortcut (preferred for this repo's tools)
|
||||
|
||||
Skip building the two inputs by hand — `scripts/probe-builtin.ts` instantiates the live tool, pulls the EXACT wire schema (`toolWireSchema`) and rendered prompt (`tool.description`), and derives the outline for you:
|
||||
|
||||
```bash
|
||||
bun .omp/skills/tool-prompt-optimization/scripts/probe-builtin.ts --tool <name> [--no-summary] [--show]
|
||||
```
|
||||
|
||||
- `--show` prints the resolved schema + derived outline + real prompt and exits (no API calls) — use it to eyeball inputs before spending tokens.
|
||||
- `--no-summary` runs the ablation (blank the summary line) directly.
|
||||
- `--samples` / `--model` / `--max-tokens` / `--json` forward to the panel; output ends with the REAL prompt so you can diff in place.
|
||||
- It bypasses the settings allowlist via the factory map, so gated tools (`irc`, `github`, …) resolve. If a tool refuses to construct (an availability gate like a missing `gh` CLI), fall back to the manual inputs below.
|
||||
|
||||
## Build the two inputs
|
||||
|
||||
**Schema** — use the *wire* schema the model actually sees, not a hand-sketch. For this repo's arktype tool schemas:
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
#!/usr/bin/env bun
|
||||
/**
|
||||
* Builtin shortcut for the prompt-inference probe.
|
||||
*
|
||||
* Instead of hand-writing a tool's JSON schema + outline, point this at a live
|
||||
* builtin tool name. It resolves the tool (direct constructor for availability-gated
|
||||
* `github`/`irc`, else the BUILTIN_TOOLS/HIDDEN_TOOLS factory map), pulls the EXACT wire
|
||||
* schema the model sees (`toolWireSchema`) and the rendered prompt (`tool.description`),
|
||||
* derives an outline by blanking section bodies, then runs the same `probe()` panel.
|
||||
*
|
||||
* Usage:
|
||||
* bun probe-builtin.ts --tool <name> [--no-summary] [--show]
|
||||
* --tool <name> builtin tool name (e.g. irc, github, read). Required.
|
||||
* --no-summary ablation: blank the one-line summary too (isolate schema-alone).
|
||||
* --show print resolved schema + outline + real prompt and exit (no API calls).
|
||||
* --samples / --model / --max-tokens / --json forwarded to probe().
|
||||
*
|
||||
* The heavy coding-agent import lives here; probe.ts stays pi-ai-only.
|
||||
*/
|
||||
import { parseArgs } from "node:util";
|
||||
import { toolWireSchema } from "@oh-my-pi/pi-ai";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { BUILTIN_TOOLS, GithubTool, HIDDEN_TOOLS, IrcTool, type Tool, type ToolFactory, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { probe } from "./probe.ts";
|
||||
|
||||
const OPEN_TAG = /^<[a-z_][\w-]*>$/i;
|
||||
const CLOSE_TAG = /^<\/[a-z_][\w-]*>$/i;
|
||||
const MD_HEADER = /^#{1,6}\s/;
|
||||
|
||||
/** Keep the summary + every section header/tag; collapse each body run to a single `...`. */
|
||||
function deriveOutline(description: string, dropSummary: boolean): string {
|
||||
const lines = description.split("\n");
|
||||
let i = 0;
|
||||
const summary: string[] = [];
|
||||
const isHeader = (l: string): boolean => {
|
||||
const t = l.trim();
|
||||
return OPEN_TAG.test(t) || CLOSE_TAG.test(t) || MD_HEADER.test(l);
|
||||
};
|
||||
while (i < lines.length && !isHeader(lines[i])) {
|
||||
summary.push(lines[i]);
|
||||
i++;
|
||||
}
|
||||
const parts: string[] = [dropSummary ? "" : summary.join("\n").trim()];
|
||||
let pendingBody = false;
|
||||
for (; i < lines.length; i++) {
|
||||
const line = lines[i];
|
||||
const t = line.trim();
|
||||
if (OPEN_TAG.test(t) || MD_HEADER.test(line)) {
|
||||
parts.push("", line);
|
||||
pendingBody = false;
|
||||
} else if (CLOSE_TAG.test(t)) {
|
||||
parts.push(line);
|
||||
pendingBody = false;
|
||||
} else if (t !== "") {
|
||||
if (!pendingBody) {
|
||||
parts.push("...");
|
||||
pendingBody = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return parts.join("\n").trim();
|
||||
}
|
||||
|
||||
async function resolveTool(name: string): Promise<Tool> {
|
||||
await Settings.init({ inMemory: true, cwd: process.cwd() });
|
||||
const settings = Settings.isolated({});
|
||||
// Enriched stub: we only read `.parameters` / `.description`, never `execute`, so the
|
||||
// availability-gated factories (irc needs a registry + agent id; github needs `gh`) can
|
||||
// still construct. Factory map bypasses the settings allowlist (`isToolAllowed`).
|
||||
const session = {
|
||||
cwd: process.cwd(),
|
||||
hasUI: false,
|
||||
getSessionFile: () => null,
|
||||
getSessionSpawns: () => "*",
|
||||
settings,
|
||||
taskDepth: 0,
|
||||
getAgentId: () => "Probe",
|
||||
agentRegistry: {},
|
||||
} as ToolSession;
|
||||
// `github`/`irc` map to `*.createIf`, which gates on external availability (gh CLI) or a
|
||||
// live agent registry. We only read `.parameters`/`.description`, so direct-construct those
|
||||
// two — keeps the probe gh-independent. Everything else goes through the factory map.
|
||||
const direct: Record<string, (s: ToolSession) => Tool> = {
|
||||
github: s => new GithubTool(s),
|
||||
irc: s => new IrcTool(s),
|
||||
};
|
||||
const key = name.toLowerCase();
|
||||
const directCtor = direct[key];
|
||||
if (directCtor) return directCtor(session);
|
||||
const factories: Record<string, ToolFactory> = { ...BUILTIN_TOOLS, ...HIDDEN_TOOLS };
|
||||
const factory = factories[key];
|
||||
if (!factory) {
|
||||
const names = Object.keys(factories).sort().join(", ");
|
||||
throw new Error(`unknown builtin tool "${name}". available: ${names}`);
|
||||
}
|
||||
const tool = await factory(session);
|
||||
if (!tool) {
|
||||
throw new Error(`tool "${name}" did not construct here — blocked by an availability gate (e.g. ssh, or a memory backend that isn't configured). Fall back to the manual --schema/--template path.`);
|
||||
}
|
||||
return tool;
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
args: Bun.argv.slice(2),
|
||||
options: {
|
||||
tool: { type: "string" },
|
||||
"no-summary": { type: "boolean" },
|
||||
show: { type: "boolean" },
|
||||
samples: { type: "string" },
|
||||
model: { type: "string" },
|
||||
"max-tokens": { type: "string" },
|
||||
json: { type: "boolean" },
|
||||
},
|
||||
allowPositionals: false,
|
||||
});
|
||||
|
||||
if (!values.tool) {
|
||||
console.error("usage: bun probe-builtin.ts --tool <name> [--no-summary] [--show] [--samples N] [--model p/id,...] [--max-tokens N] [--json]");
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const tool = await resolveTool(values.tool);
|
||||
const schema = toolWireSchema(tool);
|
||||
const realPrompt = tool.description ?? "";
|
||||
const outline = deriveOutline(realPrompt, Boolean(values["no-summary"]));
|
||||
|
||||
if (values.show) {
|
||||
console.log(`# tool: ${tool.name}\n\n## wire schema\n\`\`\`json\n${JSON.stringify(schema, null, 2)}\n\`\`\`\n`);
|
||||
console.log(`## derived outline\n${outline}\n`);
|
||||
console.log(`## real prompt (${Buffer.byteLength(realPrompt)} bytes)\n${realPrompt}`);
|
||||
return;
|
||||
}
|
||||
|
||||
const run = await probe({
|
||||
schema,
|
||||
template: outline,
|
||||
name: tool.name,
|
||||
samples: values.samples ? Number(values.samples) : undefined,
|
||||
models: values.model ? values.model.split(",").map(s => s.trim()).filter(Boolean) : undefined,
|
||||
maxTokens: values["max-tokens"] ? Number(values["max-tokens"]) : undefined,
|
||||
});
|
||||
|
||||
if (values.json) {
|
||||
console.log(JSON.stringify({ ...run, realPrompt }, null, 2));
|
||||
return;
|
||||
}
|
||||
|
||||
for (const result of run.results) {
|
||||
console.log(`\n############ ${result.model} ############`);
|
||||
result.samples.forEach((s, i) => {
|
||||
const tag = s.error ? `ERROR: ${s.error}` : s.stopReason;
|
||||
console.log(`\n----- sample ${i + 1}/${result.samples.length} [${tag}] -----`);
|
||||
console.log(s.error ? "" : s.text);
|
||||
});
|
||||
}
|
||||
console.log(`\n############ REAL PROMPT (${tool.name}) ############\n${realPrompt}`);
|
||||
}
|
||||
|
||||
if (import.meta.main) {
|
||||
await main();
|
||||
}
|
||||
Reference in New Issue
Block a user