Merge remote-tracking branch 'can1357/main' into fix/provider-onpayload-replacements

This commit is contained in:
DarkPhilosophy
2026-06-15 14:01:41 +03:00
288 changed files with 28272 additions and 8516 deletions
+17 -1
View File
@@ -57,4 +57,20 @@ runs:
- name: Install dependencies
shell: bash
run: bun install --frozen-lockfile
run: |
# The shared bun store (mounted PVC on omp-kata, actions/cache
# elsewhere) can hand a job a corrupt or partially written tarball
# when parallel jobs touch the same cache entry, so `bun install`
# aborts with `Fail extracting tarball for "<pkg>"`. That blob is
# content-hash-named, not derivable from the package name, so we
# can't evict just the bad entry — instead retry the install against
# a fresh job-local cache dir, forcing a clean re-download without
# mutating the shared store other concurrent jobs rely on. The warm
# shared cache stays the fast path; the cold retry only runs on the
# rare corruption (also covers a transient network blip on attempt 1).
if bun install --frozen-lockfile; then
exit 0
fi
echo "::warning title=bun install retry::shared bun store install failed; retrying with a clean job-local cache"
retry_cache="$(mktemp -d "${RUNNER_TEMP:-/tmp}/bun-cache-retry.XXXXXX")"
bun install --frozen-lockfile --cache-dir="$retry_cache"
+21 -1
View File
@@ -501,9 +501,29 @@ jobs:
- uses: ./.github/actions/ensure-rust-toolchain
with:
toolchain: nightly-2026-04-29
- uses: ./.github/actions/ensure-sccache
- name: Detect runner environment
id: detect
shell: bash
run: |
# $SCCACHE_BUCKET is injected only on self-hosted omp-kata pods; its
# presence selects baked sccache + shared S3. GitHub-hosted runners
# (PRs) need sccache-action to export the GHA cache URL/token into the
# step env — a bare binary install leaves SCCACHE_GHA_ENABLED set with
# no cache URL, so sccache server startup fails ("cache url for ghac
# not found"). Mirrors the build-native action's sccache wiring.
if [ -n "${SCCACHE_BUCKET:-}" ]; then
echo "on_infra=true" >> "$GITHUB_OUTPUT"
else
echo "on_infra=false" >> "$GITHUB_OUTPUT"
fi
- name: Ensure baked sccache (omp-kata)
if: steps.detect.outputs.on_infra == 'true'
uses: ./.github/actions/ensure-sccache
with:
version: "0.15.0"
- name: Setup sccache (GitHub-hosted)
if: steps.detect.outputs.on_infra == 'false'
uses: mozilla-actions/sccache-action@v0.0.10
- name: Enable sccache for cargo
# Conditional backend: self-hosted omp-kata injects a shared S3
# (RustFS) sccache via pod env; GitHub-hosted runners keep the GHA
+2 -2
View File
@@ -32,12 +32,12 @@ This repo contains multiple packages, but **`packages/coding-agent/`** is the pr
- **Class privacy**: use ES `#private` fields; leave externally accessible members bare. **No `private`/`protected`/`public` keyword on fields or methods**, except on **constructor parameter properties** where TypeScript requires it (e.g. `constructor(private readonly session: ToolSession)`).
- **Promises**: use `Promise.withResolvers()` instead of `new Promise((resolve, reject) => ...)`.
- **Prompts**: never build prompts in code (no inline strings, template literals, or concatenation). Prompts live in static `.md` files; use Handlebars for dynamic content. Import them via `import content from "./prompt.md" with { type: "text" }` — not `readFile`.
- **Worker scripts**: workers re-enter the CLI entrypoint; never spawn separate worker entry modules. `cli.ts` declares itself as the worker host at startup (`declareWorkerHostEntry()` from `@oh-my-pi/pi-utils/env`) and dispatches hidden argv selectors (`__omp_stats_sync_worker`, `__omp_tab_worker`, `__omp_js_eval_worker`, `--tiny-worker`) before loading the command registry. Spawn sites use:
- **Worker scripts**: workers re-enter the CLI entrypoint; never spawn separate worker entry modules. `cli.ts` declares itself as the worker host at startup (`declareWorkerHostEntry()` from `@oh-my-pi/pi-utils/env`) and dispatches hidden argv selectors (`__omp_worker_stats_sync`, `__omp_worker_tab`, `__omp_worker_js_eval`, `__omp_worker_tiny_inference`) before loading the command registry. Spawn sites use:
```ts
import { workerHostEntry } from "@oh-my-pi/pi-utils";
const hostEntry = workerHostEntry();
const worker = hostEntry
? new Worker(hostEntry, { type: "module", argv: ["__omp_<name>_worker"] })
? new Worker(hostEntry, { type: "module", argv: ["__omp_worker_<name>"] })
: new Worker(new URL("./<worker>.ts", import.meta.url).href, { type: "module" });
```
When the process was started from the omp CLI — source `cli.ts`, npm-bundle `dist/cli.js`, or compiled binary — `workerHostEntry()` is `Bun.main` and the worker re-enters the single entry module, so no per-worker `--compile` entrypoints or bundle entries exist. Outside a CLI host (`bun test`, SDK embedding, standalone `omp-stats`) it returns `null` and the direct-module fallback loads the worker source. New worker kinds MUST add their selector to the dispatch table in `cli.ts` and keep the fallback branch.
Generated
+8 -8
View File
@@ -265,9 +265,9 @@ dependencies = [
[[package]]
name = "bon"
version = "3.9.2"
version = "3.9.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b2f04f6fef12d70d42a77b1433c9e0f065238479a6cefc4f5bab105e9873a3c3"
checksum = "a602c73c7b0148ec6d12af6fd5cc7a46e2eacc8878271a999abac56eed12f561"
dependencies = [
"bon-macros",
"rustversion",
@@ -275,9 +275,9 @@ dependencies = [
[[package]]
name = "bon-macros"
version = "3.9.2"
version = "3.9.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7d0bd4c2f75335ad98052a37efb54f428b492f64340257143b3429c8a508fa7b"
checksum = "6dee98b0db6a962de883bf5d20362dee4d7ca0d12fe39a7c6c73c844e1cd7c1f"
dependencies = [
"darling 0.23.0",
"ident_case",
@@ -2330,7 +2330,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "15.13.1"
version = "15.13.3"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -2398,7 +2398,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "15.13.1"
version = "15.13.3"
dependencies = [
"async-trait",
"libc",
@@ -2410,7 +2410,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "15.13.1"
version = "15.13.3"
dependencies = [
"anyhow",
"arboard",
@@ -2458,7 +2458,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "15.13.1"
version = "15.13.3"
dependencies = [
"anyhow",
"brush-builtins",
+1 -1
View File
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
resolver = "3"
[workspace.package]
version = "15.13.1"
version = "15.13.3"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+28 -32
View File
@@ -20,7 +20,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -37,7 +37,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -51,7 +51,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -64,7 +64,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "15.13.1",
"version": "15.13.3",
"bin": {
"omp": "src/cli.ts",
},
@@ -130,7 +130,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"diff": "catalog:",
"lru-cache": "catalog:",
@@ -141,7 +141,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "15.13.1",
"version": "15.13.3",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -167,7 +167,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "15.13.1",
"version": "15.13.3",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -175,7 +175,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
@@ -187,7 +187,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "15.13.1",
"version": "15.13.3",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -213,7 +213,7 @@
},
"packages/swarm-extension": {
"name": "@oh-my-pi/swarm-extension",
"version": "15.13.1",
"version": "15.13.3",
"bin": {
"omp-swarm": "src/cli.ts",
},
@@ -229,7 +229,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -270,7 +270,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "15.13.1",
"version": "15.13.3",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"beautiful-mermaid": "catalog:",
@@ -284,7 +284,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "15.13.1",
"version": "15.13.3",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -320,18 +320,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "15.13.1",
"@oh-my-pi/omp-stats": "15.13.1",
"@oh-my-pi/pi-agent-core": "15.13.1",
"@oh-my-pi/pi-ai": "15.13.1",
"@oh-my-pi/pi-catalog": "15.13.1",
"@oh-my-pi/pi-coding-agent": "15.13.1",
"@oh-my-pi/pi-mnemopi": "15.13.1",
"@oh-my-pi/pi-natives": "15.13.1",
"@oh-my-pi/pi-tui": "15.13.1",
"@oh-my-pi/pi-utils": "15.13.1",
"@oh-my-pi/pi-wire": "15.13.1",
"@oh-my-pi/snapcompact": "15.13.1",
"@oh-my-pi/hashline": "15.13.3",
"@oh-my-pi/omp-stats": "15.13.3",
"@oh-my-pi/pi-agent-core": "15.13.3",
"@oh-my-pi/pi-ai": "15.13.3",
"@oh-my-pi/pi-catalog": "15.13.3",
"@oh-my-pi/pi-coding-agent": "15.13.3",
"@oh-my-pi/pi-mnemopi": "15.13.3",
"@oh-my-pi/pi-natives": "15.13.3",
"@oh-my-pi/pi-tui": "15.13.3",
"@oh-my-pi/pi-utils": "15.13.3",
"@oh-my-pi/pi-wire": "15.13.3",
"@oh-my-pi/snapcompact": "15.13.3",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -681,7 +681,7 @@
"@napi-rs/wasm-tools-win32-x64-msvc": ["@napi-rs/wasm-tools-win32-x64-msvc@1.0.1", "", { "os": "win32", "cpu": "x64" }, "sha512-rEAf05nol3e3eei2sRButmgXP+6ATgm0/38MKhz9Isne82T4rPIMYsCIFj0kOisaGeVwoi2fnm7O9oWp5YVnYQ=="],
"@nodable/entities": ["@nodable/entities@2.1.1", "", {}, "sha512-Pig3HxDIoMgjdEH8OCf/dkcTmLFjJRjWuq8jSnklu284/TKOPibSRERmOykiwmyXTtv61mP+44f3GMx0tLAyjg=="],
"@nodable/entities": ["@nodable/entities@2.2.0", "", {}, "sha512-9uGyhaQavEUMC8AIddIjau4NsnsXhou+j5sBAGojCM1oxmQpVKTWR/9JxABD6UAv12vpIms55fPZKFQEhG6uBg=="],
"@octokit/auth-token": ["@octokit/auth-token@6.0.0", "", {}, "sha512-P4YJBPdPSpWTQ1NU4XYdvHvXJJDxM6YwpS0FZHRgP7YFkdVxsWcpWGy/NVqlAA7PcPCnMacXlRm1y2PFZRWL/w=="],
@@ -1031,7 +1031,7 @@
"es-errors": ["es-errors@1.3.0", "", {}, "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw=="],
"es-toolkit": ["es-toolkit@1.47.0", "", {}, "sha512-n1GuoD0WEQZMBk5tttoZSqwgyLx01oqa5XsBmCHwPyNe1S9jPBEmtR2pSgp2kJuWE3ciFZ6yRHmY4pM4C3OOkw=="],
"es-toolkit": ["es-toolkit@1.47.1", "", {}, "sha512-5RAqEwf4P4E17p+W75KLOWw/nOvKZzSQpxM32IpI2KZLaVonjTrZ0Ai5ghMaVI9eKC2p8eoQgcBdkEDgzFk6+Q=="],
"es6-error": ["es6-error@4.1.1", "", {}, "sha512-Um/+FxMr9CISWh0bi5Zv0iOD+4cFh5qLeks1qhAopKVAJw3drgKbKySikp7wGhDL0HPeaja0P5ULZrxLkniUVg=="],
@@ -1357,7 +1357,7 @@
"string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="],
"string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="],
"string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="],
"strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="],
@@ -1535,8 +1535,6 @@
"string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
"string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="],
"wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="],
"xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="],
@@ -1559,8 +1557,6 @@
"fastembed/onnxruntime-node/tar": ["tar@7.5.16", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w=="],
"jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="],
"log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="],
"log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="],
+1 -1
View File
@@ -172,7 +172,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV15_13_1")]
#[napi(js_name = "__piNativesV15_13_3")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+1
View File
@@ -151,6 +151,7 @@ Default prune policy:
- Protect newest `40_000` tool-output tokens.
- Require at least `20_000` total estimated savings.
- Never blank a result below `50` tokens (`MIN_PRUNE_TOKENS`): the `[Output truncated - N tokens]` placeholder costs ~8 tokens, so pruning a sub-floor result would grow the context and churn the prompt cache for nothing. (Superseded and useless results keep their own rules — the useless collector already drops no-savings candidates; superseded reads prune for correctness regardless of size.)
- Never prune `skill` tool results, `read` results of `skill://` paths, or reads of the active plan reference file (added via `AgentSession`'s plan protection).
Pruned tool results are replaced with:
+630
View File
@@ -0,0 +1,630 @@
# Anthropic Claude tool use (Messages API content blocks)
Anthropic's Claude is a closed, hosted model family; there are no released weights and therefore no `--tool-call-parser` flag to set. The canonical tool-calling convention is the **Messages API** (`POST /v1/messages`, header `anthropic-version: 2023-06-01`): tools are advertised in a top-level `tools` array, the model returns structured `tool_use` **content blocks** with `stop_reason: "tool_use"`, and you feed results back as `tool_result` content blocks inside a `user` message. Tool use is "enabled" simply by including the `tools` parameter (optionally with `tool_choice`); the API then injects a tool-use system prompt and parses the model's output back into JSON blocks for you. This applies to all current models (Claude Opus / Sonnet / Haiku 3.x, 4, 4.x) and is mirrored by gateways such as LiteLLM and by third-party Claude-compatible servers.
Under the hood the model is trained to emit an **XML** function-call syntax (`<function_calls>` / `<invoke>` / `<parameter>`); the API serializes your JSON-Schema tools into a system prompt and converts the model's XML output into JSON `tool_use` blocks. That underlying format is documented as the *secondary* convention below, together with the older, now-retired prompt-based **legacy XML** format (`<tool_name>` / `<parameters>` / `<function_results>`) that pre-dates the Messages API and still surfaces when you do tool use purely through prompting.
The primary, authoritative shape for any parser/renderer is the JSON content-block format. The XML is informational (and the only thing visible if you reconstruct prompts at the token level).
---
## Content-block types & stop reasons
Anthropic has no token-level tool delimiters in the public API. The unit is the **content block**: every `message.content` is an array of typed blocks. Tool calling adds two block types and one stop reason; streaming adds a delta type.
| Item | Where | Shape / meaning |
| --- | --- | --- |
| `text` block | assistant & user | `{"type":"text","text":"..."}`. Plain prose. Assistant may emit text *before* its tool calls. |
| `tool_use` block | assistant | `{"type":"tool_use","id":"toolu_...","name":"<tool>","input":{...}}`. The function call. `input` is a **nested JSON object** (already parsed), conforming to the tool's `input_schema`. |
| `tool_result` block | user | `{"type":"tool_result","tool_use_id":"toolu_...","content":<string \| block[]>,"is_error":<bool?>}`. The executed result, sent back in a `user` message. |
| `server_tool_use` block | assistant | `{"type":"server_tool_use","id":"srvtoolu_...","name":"web_search","input":{...}}`. Emitted for Anthropic-executed server tools; you do **not** return a `tool_result` for these. |
| `web_search_tool_result` (and similar) | assistant | Server-tool output, injected by Anthropic inline in the assistant turn. |
| `thinking` / `redacted_thinking` block | assistant | Extended-thinking reasoning blocks; carry a `signature`. Must be preserved verbatim across turns when thinking + tools are combined. |
| `stop_reason: "tool_use"` | response top level | The model invoked one or more tools and is waiting for results. Drives the agentic loop. |
| `stop_reason: "end_turn"` | response top level | Natural completion (no tool call); the loop exits. |
| Other `stop_reason` | response top level | `"max_tokens"`, `"stop_sequence"`, `"pause_turn"` (long server-tool turn, resend as-is to continue), `"refusal"`. |
| `id` prefixes | — | Messages `msg_…`; client tool calls `toolu_…`; server tool calls `srvtoolu_…`. |
Streaming adds these SSE events / delta types (full list under [Roles / channels](#roles--channels--turn-structure) and [Tool-call format](#tool-call-format)):
| Streaming item | Shape / meaning |
| --- | --- |
| `message_start` | Carries a `Message` skeleton with empty `content`, `stop_reason: null`. |
| `content_block_start` | Opens a block at `index`. For a tool call: `content_block.{type:"tool_use",id,name,input:{}}` — `input` starts as an **empty object**. |
| `content_block_delta` / `input_json_delta` | `{"type":"input_json_delta","partial_json":"<chunk>"}` — a **partial JSON string** fragment of `tool_use.input`. |
| `content_block_delta` / `text_delta` | `{"type":"text_delta","text":"..."}`. |
| `content_block_delta` / `thinking_delta`, `signature_delta` | Extended-thinking content / signature. |
| `content_block_stop` | Closes the block at `index`; this is when accumulated `partial_json` is complete and safe to `JSON.parse`. |
| `message_delta` | Top-level updates; carries the final `delta.stop_reason` (e.g. `"tool_use"`) and **cumulative** `usage`. |
| `message_stop` | End of stream. |
| `ping` / `error` | Keep-alive; `error` (e.g. `overloaded_error`) may appear mid-stream. |
### Legacy XML tags (prompt-based, pre-Messages-API)
The retired prompt-based format used these tags. They are nested-element tags (no attributes), distinct from the modern attribute form (`<invoke name="…">`). Verified against Anthropic's archived "Legacy tool use" doc (see [Sources](#sources)).
| Tag | Role | Notes |
| --- | --- | --- |
| `<tools>` … `</tools>` | tool advertising | Container in the system prompt wrapping all `<tool_description>` entries. |
| `<tool_description>` | tool advertising | One per tool: holds `<tool_name>`, `<description>`, `<parameters>`. |
| `<tool_name>` | both | Function name (used in definitions, calls, and results). |
| `<parameters>` / `<parameter>` | definition | `<parameters>` wraps `<parameter>` entries, each with `<name>`, `<type>`, `<description>`. |
| `<function_calls>` | model output | Wraps one or more `<invoke>` blocks. |
| `<invoke>` | model output | One function call; contains `<tool_name>` + a `<parameters>` block of `<paramName>value</paramName>` child tags. |
| `<function_results>` | tool result (fed back) | Wraps `<result>` (success) or `<error>` (failure). |
| `<result>` / `<stdout>` | tool result | `<result>` holds `<tool_name>` + `<stdout>`; the output text goes in `<stdout>`. |
| `<error>` | tool result | Replaces `<result>` when the function raised. |
| `</function_calls>` | stop sequence | Passed as `stop_sequence` so generation halts after a call. |
| `<scratchpad>` / `<answer>` | model output | Conventionally used for chain-of-thought and final answer in legacy prompts. |
---
## Roles / channels / turn structure
The Messages API uses only two conversational roles, `user` and `assistant`, alternating. There is **no** dedicated `tool`/`function` role and **no** top-level `system` role — the system prompt is a separate top-level `system` parameter (string or text-block array). Tool data rides inside the normal roles:
- `assistant` messages contain AI-generated `text`, `thinking`, and `tool_use` (and `server_tool_use`) blocks.
- `user` messages contain your `text`/`image`/`document` content and `tool_result` blocks.
There are no named "channels". The closest analogue to a reasoning channel is the extended-thinking `thinking` content block (a first-class block with a cryptographic `signature`), kept separate from the user-visible `text` block. When thinking is enabled alongside tools, the `thinking` block(s) from a tool-calling turn must be passed back unmodified in the follow-up request.
The agentic loop is keyed on `stop_reason`:
1. Send `tools` + the user message.
2. Claude responds with `stop_reason: "tool_use"` and one or more `tool_use` blocks (optionally preceded by a `text` block).
3. Execute each tool; build a `tool_result` block per call.
4. Append the assistant message **and** a `user` message carrying all `tool_result` blocks; resend.
5. Repeat while `stop_reason == "tool_use"`; exit on `end_turn` (or another terminal reason).
Strict ordering rules (a 400 otherwise):
- `tool_result` blocks must come **first** in the `user` message's `content` array (any text after them).
- The `tool_result` `user` message must **immediately follow** the assistant `tool_use` message — nothing in between.
- Every `tool_use.id` must be answered by a `tool_result.tool_use_id` in that next message.
---
## Tool definitions
Tools are passed in the top-level `tools` array. Each user-defined (client) tool is a **flat** object — no `{"type":"function", "function":{…}}` wrapper (that wrapper is OpenAI's). Fields:
- `name` — matches `^[a-zA-Z0-9_-]{1,64}$`.
- `description` — detailed plaintext (the single biggest driver of tool-call quality).
- `input_schema` — a JSON Schema object (**not** `parameters`) describing the input the model must produce.
- Optional: `input_examples`, `cache_control`, `strict`, `defer_loading`, `allowed_callers`.
```json
{
"name": "get_weather",
"description": "Get the current weather in a given location",
"input_schema": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA"
},
"unit": {
"type": "string",
"enum": ["celsius", "fahrenheit"],
"description": "The unit of temperature, either 'celsius' or 'fahrenheit'"
}
},
"required": ["location"]
}
}
```
Anthropic-schema client tools (`bash`, `text_editor`, `computer`, `memory`) and server tools (`web_search`, `web_fetch`, `code_execution`, `tool_search`) instead carry a versioned `type`, e.g. `{"type": "web_search_20250305", "name": "web_search"}`.
`tool_choice` controls invocation (four options):
- `{"type":"auto"}` — model decides (default when `tools` present).
- `{"type":"any"}` — must call some tool.
- `{"type":"tool","name":"get_weather"}` — must call that specific tool.
- `{"type":"none"}` — no tools (default when no `tools`).
With `any` or `tool` the API prefills the assistant turn, so no leading natural-language text precedes the `tool_use` block. Add `"disable_parallel_tool_use": true` inside `tool_choice` to cap at one tool per turn. (Extended thinking only supports `auto`/`none`.)
### How the API turns this into a prompt (the bridge to XML)
When `tools` is present, the API constructs a tool-use system prompt with this skeleton (verified from "Define tools"):
```text
In this environment you have access to a set of tools you can use to answer the user's question.
{{ FORMATTING INSTRUCTIONS }}
String and scalar parameters should be specified as is, while lists and objects should use JSON format. Note that spaces for string values are not stripped. The output is not expected to be valid XML and is parsed with regular expressions.
Here are the functions available in JSONSchema format:
{{ TOOL DEFINITIONS IN JSON SCHEMA }}
{{ USER SYSTEM PROMPT }}
{{ TOOL CONFIGURATION }}
```
`{{ TOOL DEFINITIONS IN JSON SCHEMA }}` is your `tools` array serialized to JSON Schema. `{{ FORMATTING INSTRUCTIONS }}` is the (unpublished) block teaching the model the XML syntax with `antml:` namespace prefixes (shown under [Tool-call format → underlying XML](#underlying-xml-modern-attribute-form-with-antml-namespace)). The note "parsed with regular expressions" is why output need not be well-formed XML.
---
## Tool-call format
The wire format your application consumes is JSON. A single call is one `tool_use` content block in the assistant message, with `stop_reason: "tool_use"` at the top level:
```json
{
"id": "msg_01Aq9w938a90dw8q",
"type": "message",
"role": "assistant",
"model": "claude-opus-4-8",
"content": [
{
"type": "text",
"text": "I'll check the current weather in San Francisco for you."
},
{
"type": "tool_use",
"id": "toolu_01A09q90qw90lq917835lq9",
"name": "get_weather",
"input": { "location": "San Francisco, CA", "unit": "celsius" }
}
],
"stop_reason": "tool_use",
"stop_sequence": null,
"usage": { "input_tokens": 472, "output_tokens": 65 }
}
```
Key facts for a parser:
- `tool_use.input` is an already-parsed **object**, never a JSON string.
- A leading `text` block is optional and informational; do not rely on its wording.
- Match calls to results by `id` → `tool_use_id`.
### Underlying XML (modern attribute form with antml: namespace)
Before the API converts it, the model literally emits an XML block. The current (Claude 3+) form is attribute-based:
```text
<function_calls>
<invoke name="get_weather">
<parameter name="location">San Francisco, CA</parameter>
<parameter name="unit">celsius</parameter>
</invoke>
</function_calls>
```
Current Claude models prefix these tags with an `antml:` XML namespace prefix (e.g. `antml:function_calls`, `antml:invoke name="…"`, `antml:parameter name="…"`). The API strips all of this and exposes only the JSON `tool_use` block; integrators should target the JSON, not the XML.
---
## Multiple / parallel tool calls
Parallel calls are the default. Claude emits **multiple `tool_use` blocks in a single assistant message**:
```json
{
"role": "assistant",
"content": [
{ "type": "text", "text": "Let me check both cities." },
{
"type": "tool_use",
"id": "toolu_01weather_sf",
"name": "get_weather",
"input": { "location": "San Francisco, CA" }
},
{
"type": "tool_use",
"id": "toolu_02weather_nyc",
"name": "get_weather",
"input": { "location": "New York, NY" }
}
]
}
```
You return **all** results in **one** `user` message, one `tool_result` per call, results first:
```json
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01weather_sf",
"content": "San Francisco: 68F, partly cloudy"
},
{
"type": "tool_result",
"tool_use_id": "toolu_02weather_nyc",
"content": "New York: 45F, clear skies"
}
]
}
```
Calls in one turn are **unordered** and may be run concurrently. If two batched calls turn out to depend on each other, return the natural error in a `tool_result` with `"is_error": true`; Claude reissues the dependent call on a later turn. (In the legacy XML format, parallelism is multiple `<invoke>` blocks inside one `<function_calls>`.)
---
## Tool-result format
A result is a `tool_result` block inside a `user` message:
- `tool_use_id` (required) — the `id` of the `tool_use` it answers.
- `content` (optional) — a string, **or** an array of `text`/`image`/`document` blocks. Omit for an empty result.
- `is_error` (optional) — `true` for execution failures; put a useful message in `content`.
```json
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01A09q90qw90lq917835lq9",
"content": "15 degrees"
}
]
}
```
Error result:
```json
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01A09q90qw90lq917835lq9",
"content": "ConnectionError: the weather service API is not available (HTTP 500)",
"is_error": true
}
]
}
```
Rich result (text + image blocks):
```json
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01A09q90qw90lq917835lq9",
"content": [
{ "type": "text", "text": "15 degrees" },
{
"type": "image",
"source": { "type": "base64", "media_type": "image/jpeg", "data": "/9j/4AAQSkZJRg..." }
}
]
}
]
}
```
Server tools require **no** `tool_result` from you — Anthropic executes them and injects the result inline in the assistant turn. (Legacy XML feeds results back as `<function_results><result><tool_name>…</tool_name><stdout>…</stdout></result></function_results>`, or `<error>…</error>` on failure.)
---
## End-to-end example
A complete multi-turn weather exchange. All JSON is valid.
**Request 1 — system + tools + user question:**
```json
{
"model": "claude-opus-4-8",
"max_tokens": 1024,
"system": "You are a helpful weather assistant. Use the provided tools to answer.",
"tools": [
{
"name": "get_weather",
"description": "Get the current weather in a given location",
"input_schema": {
"type": "object",
"properties": {
"location": { "type": "string", "description": "The city and state, e.g. San Francisco, CA" },
"unit": { "type": "string", "enum": ["celsius", "fahrenheit"], "description": "Unit for the temperature" }
},
"required": ["location"]
}
}
],
"messages": [
{ "role": "user", "content": "What's the weather in San Francisco?" }
]
}
```
**Response 1 — assistant requests the tool (`stop_reason: "tool_use"`):**
```json
{
"id": "msg_01Aq9w938a90dw8q",
"type": "message",
"role": "assistant",
"model": "claude-opus-4-8",
"content": [
{ "type": "text", "text": "I'll check the current weather in San Francisco for you." },
{
"type": "tool_use",
"id": "toolu_01A09q90qw90lq917835lq9",
"name": "get_weather",
"input": { "location": "San Francisco, CA", "unit": "celsius" }
}
],
"stop_reason": "tool_use",
"stop_sequence": null,
"usage": { "input_tokens": 472, "output_tokens": 65 }
}
```
**Request 2 — replay history, append the assistant turn and the `tool_result`:**
```json
{
"model": "claude-opus-4-8",
"max_tokens": 1024,
"system": "You are a helpful weather assistant. Use the provided tools to answer.",
"tools": [
{
"name": "get_weather",
"description": "Get the current weather in a given location",
"input_schema": {
"type": "object",
"properties": {
"location": { "type": "string", "description": "The city and state, e.g. San Francisco, CA" },
"unit": { "type": "string", "enum": ["celsius", "fahrenheit"], "description": "Unit for the temperature" }
},
"required": ["location"]
}
}
],
"messages": [
{ "role": "user", "content": "What's the weather in San Francisco?" },
{
"role": "assistant",
"content": [
{ "type": "text", "text": "I'll check the current weather in San Francisco for you." },
{
"type": "tool_use",
"id": "toolu_01A09q90qw90lq917835lq9",
"name": "get_weather",
"input": { "location": "San Francisco, CA", "unit": "celsius" }
}
]
},
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01A09q90qw90lq917835lq9",
"content": "15 degrees Celsius, partly cloudy"
}
]
}
]
}
```
**Response 2 — assistant's final answer (`stop_reason: "end_turn"`):**
```json
{
"id": "msg_01EeFG3hijk2lmno4PqrSt",
"type": "message",
"role": "assistant",
"model": "claude-opus-4-8",
"content": [
{ "type": "text", "text": "It's currently 15 degrees Celsius and partly cloudy in San Francisco." }
],
"stop_reason": "end_turn",
"stop_sequence": null,
"usage": { "input_tokens": 530, "output_tokens": 18 }
}
```
### Streaming (SSE) shape of the tool call
The same tool call, streamed. Note `tool_use` opens with an empty `input`, the arguments arrive as `input_json_delta.partial_json` fragments, and the final `stop_reason` lands in `message_delta`. This block is reproduced verbatim from Anthropic's streaming docs:
```text
event: message_start
data: {"type":"message_start","message":{"id":"msg_014p7gG3wDgGV9EUtLvnow3U","type":"message","role":"assistant","model":"claude-opus-4-8","stop_sequence":null,"usage":{"input_tokens":472,"output_tokens":2},"content":[],"stop_reason":null}}
event: content_block_start
data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}
event: ping
data: {"type": "ping"}
event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Okay"}}
event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" let"}}
event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"'s"}}
event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" check"}}
event: content_block_stop
data: {"type":"content_block_stop","index":0}
event: content_block_start
data: {"type":"content_block_start","index":1,"content_block":{"type":"tool_use","id":"toolu_01T1x1fJ34qAmk2tNTrN7Up6","name":"get_weather","input":{}}}
event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":""}}
event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"{\"location\":"}}
event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" \"San"}}
event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" Francisc"}}
event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"o,"}}
event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" CA\"}"}}
event: content_block_stop
data: {"type":"content_block_stop","index":1}
event: message_delta
data: {"type":"message_delta","delta":{"stop_reason":"tool_use","stop_sequence":null},"usage":{"output_tokens":89}}
event: message_stop
data: {"type":"message_stop"}
```
Reassembly: concatenate every `partial_json` for a given `index` (`"" + "{\"location\":" + " \"San" + " Francisc" + "o," + " CA\"}"` → `{"location": "San Francisco, CA"}`), then `JSON.parse` at that block's `content_block_stop`. Tool use also supports fine-grained streaming (`eager_input_streaming` per tool) for finer `partial_json` chunking.
---
## OpenAI-compatible API mapping
Anthropic integrates tools into the `user`/`assistant` message structure rather than using OpenAI's separate `tool` role and `function` wrapper. Field-by-field:
| Concept | Anthropic Messages API | OpenAI Chat Completions |
| --- | --- | --- |
| Tool definition wrapper | flat `{"name","description","input_schema"}` in `tools[]` | `{"type":"function","function":{"name","description","parameters"}}` in `tools[]` |
| Tool schema key | `input_schema` (JSON Schema) | `parameters` (JSON Schema) |
| "Must call a tool" | `tool_choice:{"type":"any"}` / `{"type":"tool","name":…}` | `tool_choice:"required"` / `{"type":"function","function":{"name":…}}` |
| Disable parallel calls | `tool_choice:{…,"disable_parallel_tool_use":true}` | `parallel_tool_calls:false` (top level) |
| Assistant call container | `tool_use` **content block** in `content[]` | `tool_calls[]` on the assistant `message` |
| Call id | `tool_use.id` = `toolu_…` | `tool_calls[].id` = `call_…` |
| Function name | `tool_use.name` | `tool_calls[].function.name` |
| Function arguments | `tool_use.input` = **nested JSON object** (parsed) | `tool_calls[].function.arguments` = **JSON string** (must `JSON.parse`) |
| "Tools were called" signal | `stop_reason:"tool_use"` | `finish_reason:"tool_calls"` |
| Result message role | `user` message containing `tool_result` block(s) | dedicated `{"role":"tool",…}` message(s) |
| Result ↔ call linkage | `tool_result.tool_use_id` | `tool` message `tool_call_id` |
| Result payload | `tool_result.content` = string **or** block array (text/image/document) | `tool` message `content` = string |
| Error result | `tool_result` with `is_error:true` | no dedicated flag; encode in `content` |
| System prompt | top-level `system` param (no `system` role) | `{"role":"system",…}` message |
| Streamed args | `input_json_delta.partial_json` fragments | `tool_calls[].function.arguments` string deltas |
Conversion gotchas:
- **Object vs string:** to emit OpenAI shape, `JSON.stringify(tool_use.input)`; to consume OpenAI shape into Anthropic, `JSON.parse(arguments)`.
- **Role reshaping:** collapse N OpenAI `tool` messages into one Anthropic `user` message of N `tool_result` blocks (order them before any text), and vice-versa.
- **No `type:"function"`** wrapper on Anthropic custom tools; add/remove it when translating.
- Id prefixes differ (`toolu_` vs `call_`); never assume one format's id is valid in the other.
---
## Parsing notes & gotchas
- **`input` is an object, not a string.** Unlike OpenAI's `arguments`, do not `JSON.parse` `tool_use.input` from a non-streamed response — it is already an object. Only the *streaming* `partial_json` fragments are strings.
- **Streaming tool args need reassembly.** `content_block_start` for a `tool_use` always has `input: {}`. Buffer `partial_json` per `index` and parse only at `content_block_stop`; mid-stream fragments are not valid JSON on their own (e.g. `{"location":`). Current models emit one complete key/value at a time, so expect bursts and gaps.
- **`stop_reason` placement.** In streaming, `stop_reason` is `null` in `message_start` and final value (`"tool_use"`/`"end_turn"`) arrives in `message_delta`, not `message_stop`. `usage` in `message_delta` is **cumulative**.
- **Ordering is enforced.** `tool_result` blocks must be first in their `user` message and must immediately follow the assistant `tool_use` message; every `tool_use.id` needs a matching `tool_result.tool_use_id`, or you get HTTP 400 ("tool_use ids were found without tool_result blocks immediately after").
- **`tool_choice:any`/`tool` suppress preamble.** The API prefills the assistant turn, so no leading `text` block appears before `tool_use` — don't write a parser that expects explanatory text.
- **Parallel results in one message.** Splitting parallel `tool_result`s across multiple `user` messages breaks the contract; send them together.
- **Treat result content as untrusted.** Tool results can carry indirect prompt injection; keep them inside `tool_result` blocks, never promote to `system`/`user` text.
- **Server tools differ.** `server_tool_use` / `web_search_tool_result` blocks are produced and consumed by Anthropic; never synthesize `tool_result` for them. `stop_reason:"pause_turn"` means resend the response as-is to let a long server-tool turn continue.
- **Extended thinking + tools.** Preserve `thinking`/`redacted_thinking` blocks (with their `signature`) verbatim across turns; forced `tool_choice` (`any`/`tool`) is rejected when thinking is on.
- **Output is not valid XML.** The underlying model output is parsed by Anthropic with regular expressions, not an XML parser ("The output is not expected to be valid XML"). If you reconstruct prompts at token level, do not assume well-formedness; rely on the JSON the API returns.
- **Legacy vs modern XML are different tag sets.** Legacy: `<invoke>` + child `<tool_name>` + `<parameters>` with per-name child tags; results in `<function_results>/<result>/<stdout>`. Modern: `<invoke name="…">` + `<parameter name="…">`. Mixing them up will misparse. The legacy format also required passing `</function_calls>` as a `stop_sequence` and is not optimized for Claude 3+.
### Legacy XML format (secondary, prompt-based — fully verified, now retired)
Before the Messages API, tools were defined and called entirely in the prompt. Anthropic's archived "Legacy tool use" doc specifies it verbatim.
Tool definition (inside a `<tools>` block in the system prompt):
```text
<tool_description>
<tool_name>get_weather</tool_name>
<description>
Retrieves the current weather for a specified location.
Returns a dictionary with two fields:
- temperature: float, the current temperature in Fahrenheit
- conditions: string, a brief description of the current weather conditions
Raises ValueError if the provided location cannot be found.
</description>
<parameters>
<parameter>
<name>location</name>
<type>string</type>
<description>The city and state, e.g. San Francisco, CA</description>
</parameter>
</parameters>
</tool_description>
```
Model-emitted call (multiple `<invoke>` for parallel calls; pass `</function_calls>` as a `stop_sequence`):
```text
<function_calls>
<invoke>
<tool_name>get_weather</tool_name>
<parameters>
<location>San Francisco, CA</location>
</parameters>
</invoke>
</function_calls>
```
Result fed back into the next user turn:
```text
<function_results>
<result>
<tool_name>get_weather</tool_name>
<stdout>
59 degrees Fahrenheit, partly cloudy
</stdout>
</result>
</function_results>
```
Error result:
```text
<function_results>
<error>
error message goes here
</error>
</function_results>
```
The legacy system-prompt preamble (verbatim from the archived doc) was:
```text
In this environment you have access to a set of tools you can use to answer the user's question.
You may call them like this:
<function_calls>
<invoke>
<tool_name>$TOOL_NAME</tool_name>
<parameters>
<$PARAMETER_NAME>$PARAMETER_VALUE</$PARAMETER_NAME>
...
</parameters>
</invoke>
</function_calls>
Here are the tools available:
<tools>
...one <tool_description> per tool...
</tools>
```
Legacy notes: no built-in tools (everything is prompt-defined); Anthropic recommended ≤3–5 tools; the model conventionally wrapped reasoning in `<scratchpad>` and final output in `<answer>`. This format is "out of date" and "not optimized for Claude 3" — use the JSON Messages API for anything current.
---
## Sources
- Tool use overview — https://docs.claude.com/en/docs/agents-and-tools/tool-use/overview
- How tool use works — https://docs.claude.com/en/docs/agents-and-tools/tool-use/how-tool-use-works
- Define tools (tool schema, `input_schema`, `tool_choice`, constructed system prompt) — https://docs.claude.com/en/docs/agents-and-tools/tool-use/define-tools
- Handle tool calls (`tool_use`/`tool_result`, `is_error`, ordering rules) — https://docs.claude.com/en/docs/agents-and-tools/tool-use/handle-tool-calls
- Parallel tool use — https://docs.claude.com/en/docs/agents-and-tools/tool-use/parallel-tool-use
- Streaming messages (SSE events, `input_json_delta`, verbatim tool-use stream) — https://docs.claude.com/en/docs/build-with-claude/streaming
- Messages API reference (`stop_reason` enum, response shape, `tools`) — https://docs.claude.com/en/api/messages
- Legacy tool use (archived; verbatim XML tags and prompt) — https://web.archive.org/web/20240528231249/https://docs.anthropic.com/en/docs/legacy-tool-use ; also live localized copies, e.g. https://docs.anthropic.com/de/docs/legacy-tool-use (English path now redirects to the tool-use overview)
+356
View File
@@ -0,0 +1,356 @@
# DeepSeek tool-calling wire format
DeepSeek's chat models (DeepSeek-V3, V3-0324, R1, R1-0528, and DeepSeek-V3.1) share a
single tokenizer family and a distinctive envelope built from **fullwidth-pipe** special
tokens such as `<|begin▁of▁sentence|>` and `<|User|>`. Tool calling is emitted as a run
of dedicated special tokens (`<|tool▁calls▁begin|>` … `<|tool▁calls▁end|>`) rather than
JSON-in-text or XML. This document centers on **DeepSeek-V3.1** (the current hybrid
thinking/non-thinking model) and documents the older **DeepSeek-V3-0324** and
**DeepSeek-R1-0528** format as an explicit version difference, because their on-the-wire
tool syntax is *not* the same as V3.1's.
An inference server enables it with a chat template plus a tool-call parser:
- vLLM V3.1: `--enable-auto-tool-choice --tool-call-parser deepseek_v31 --chat-template examples/tool_chat_template_deepseekv31.jinja` (optionally `--reasoning-parser deepseek_r1`).
- vLLM V3-0324 / R1-0528: `--enable-auto-tool-choice --tool-call-parser deepseek_v3 --chat-template examples/tool_chat_template_deepseekv3.jinja` (V3-0324) or `tool_chat_template_deepseekr1.jinja` (R1-0528).
- The model's own `tokenizer_config.json` `chat_template` (and the identical `assets/chat_template.jinja`) renders the V3.1 envelope, tool calls, and tool outputs; it does **not** synthesize the `## Tools` advertisement block, so vLLM ships a template that does (see below).
> Verified against: the DeepSeek-V3.1 model card "Chat Template" / "ToolCall" sections, the
> byte-identical `chat_template` in `tokenizer_config.json` and `assets/chat_template.jinja`,
> the `added_tokens` in `tokenizer.json` (token IDs), `config.json` (bos/eos IDs), the
> DeepSeek-V3-0324 and DeepSeek-R1-0528 `tokenizer_config.json` chat templates, the vLLM
> `tool_chat_template_deepseekv31.jinja`, and the vLLM tool-calling / reasoning-outputs docs.
## A note on the unusual Unicode (do not substitute ASCII)
DeepSeek's markers do **not** use the ASCII vertical bar `|` (U+007C) or ASCII underscore
`_`. They use:
- `|` — **U+FF5C FULLWIDTH VERTICAL LINE**, as the delimiter just inside the angle brackets.
- `▁` — **U+2581 LOWER ONE EIGHTH BLOCK** (the SentencePiece word-boundary glyph), as the
separator *between words* inside a token, e.g. `begin▁of▁sentence`, `tool▁calls▁begin`.
So `<|tool▁calls▁begin|>` is `<` + `|`(FF5C) + `tool` + `▁`(2581) + `calls` + `▁`(2581) +
`begin` + `|`(FF5C) + `>`. Copying these tokens as `<|tool_calls_begin|>` (ASCII pipe +
underscore) produces tokens the model never trained on and will silently break parsing and
generation. The only DeepSeek markers that use ASCII brackets are the thinking tags
`<think>` / `</think>` (plain `<`, `/`, `>`) and the rarely used `<|EOT|>` (ASCII pipes).
## Special tokens
Token IDs are from DeepSeek-V3.1 `tokenizer.json` (`added_tokens`); `vocab_size` is 129280.
The `special` column reflects the tokenizer's `"special"` flag (it governs
`skip_special_tokens`); note that the role/think/tool markers are `special: false`.
| Token (verbatim) | ID | `special` | Purpose |
| --- | --- | --- | --- |
| `<|begin▁of▁sentence|>` | 0 | true | BOS; prepended once at the very start of the prompt. |
| `<|end▁of▁sentence|>` | 1 | true | EOS; ends every assistant/tool turn and is the stop token. |
| `<|▁pad▁|>` | 2 | true | Padding (`pad_token`; the model card/config also reuse EOS as pad). |
| `<|search▁begin|>` | 128796 | false | Search-agent query open (thinking-mode search tool). |
| `<|search▁end|>` | 128797 | false | Search-agent query close. |
| `<think>` | 128798 | false | Opens the reasoning/thinking span. ASCII brackets. |
| `</think>` | 128799 | false | Closes the reasoning span; **also emitted in non-thinking mode** (see below). |
| `<|fim▁hole|>` / `<|fim▁begin|>` / `<|fim▁end|>` | 128800–128802 | false | Fill-in-the-middle (not chat). |
| `<|User|>` | 128803 | false | User role marker. |
| `<|Assistant|>` | 128804 | false | Assistant role marker. |
| `<\|EOT\|>` | 128805 | true | End-of-turn (legacy; ASCII pipes, rarely used in chat). |
| `<|tool▁calls▁begin|>` | 128806 | false | Opens the assistant's batch of tool calls. |
| `<|tool▁calls▁end|>` | 128807 | false | Closes the batch of tool calls. |
| `<|tool▁call▁begin|>` | 128808 | false | Opens a single tool call inside the batch. |
| `<|tool▁call▁end|>` | 128809 | false | Closes a single tool call. |
| `<|tool▁outputs▁begin|>` | 128810 | false | Opens a batch of tool results (**R1-0528 / V3-0324 only**). |
| `<|tool▁outputs▁end|>` | 128811 | false | Closes a batch of tool results (**R1-0528 / V3-0324 only**). |
| `<|tool▁output▁begin|>` | 128812 | false | Opens a single tool result. |
| `<|tool▁output▁end|>` | 128813 | false | Closes a single tool result. |
| `<|tool▁sep|>` | 128814 | false | Separator inside a tool call (between name and arguments). |
`config.json` confirms `bos_token_id: 0`, `eos_token_id: 1`.
## Roles / channels / turn structure
There is no OpenAI-style `system`/`developer` channel token. Roles are inline markers and
the prompt is one flat string:
```text
<|begin▁of▁sentence|>{system_prompt}<|User|>{query}<|Assistant|>{response}<|end▁of▁sentence|>
```
- **System prompt** has no marker. All `system` messages are concatenated (joined with
`\n\n` when there are several) and emitted immediately after `<|begin▁of▁sentence|>`,
before the first `<|User|>`. When tools are present the `## Tools` block is appended to
this system text (separated by `\n\n`).
- **User turn**: `<|User|>` + content. (No EOS after the user text in V3.1; the assistant
marker follows directly.)
- **Assistant turn**: opens with `<|Assistant|>`, then a thinking tag, then content, then
`<|end▁of▁sentence|>`.
- **Thinking vs non-thinking (V3.1 hybrid)** — selected by the template, not by the model:
- Non-thinking generation prefix: `…<|Assistant|></think>` — the model starts *after* a
`</think>` it never had to open. Unlike DeepSeek-V3, V3.1 always injects this `</think>`.
- Thinking generation prefix: `…<|Assistant|><think>` — the model emits its chain of
thought, closes with `</think>`, then the answer.
- In multi-turn context, **every** stored assistant turn keeps a `</think>`; only the last
turn's leading thinking tag reflects the requested mode. When rendering a stored
assistant message, any text up to and including `</think>` is stripped from `content`
before re-emitting (the template does `content.split('</think>', 1)[1]`).
- **Tool calling runs in non-thinking mode.** The model card states "Toolcall is supported
in non-thinking mode," and the V3.1 tool template opens the tool-call turn with
`<|Assistant|></think>`. With vLLM, V3.1 reasoning is disabled by default; enable it via
`chat_template_kwargs={"thinking": true}`.
- **Search-agent channel**: a separate thinking-mode protocol using `<|search▁begin|>` /
`<|search▁end|>` (see the model card's `assets/search_tool_trajectory.html`); out of
scope for ordinary function calling.
## Tool definitions
Tools are advertised as a **Markdown block injected into the system area** (after the system
prompt, before the first `<|User|>`). The chat template in `tokenizer_config.json` does not
build this block from a `tools=[…]` argument; the caller (or vLLM's
`tool_chat_template_deepseekv31.jinja`) constructs it. Reproduced verbatim from the
DeepSeek-V3.1 model card, the full layout is
`<|begin▁of▁sentence|>{system prompt}\n\n{tool_description}<|User|>{query}<|Assistant|></think>`
where `{tool_description}` is:
```text
## Tools
You have access to the following tools:
### {tool_name1}
Description: {description}
Parameters: {json.dumps(parameters)}
IMPORTANT: ALWAYS adhere to this exact format for tool use:
<|tool▁calls▁begin|><|tool▁call▁begin|>tool_call_name<|tool▁sep|>tool_call_arguments<|tool▁call▁end|>{additional_tool_calls}<|tool▁calls▁end|>
Where:
- `tool_call_name` must be an exact match to one of the available tools
- `tool_call_arguments` must be valid JSON that strictly follows the tool's Parameters Schema
- For multiple tool calls, chain them directly without separators or spaces
```
Each tool contributes one `### {name}` section with a `Description:` line and a
`Parameters: {…}` line whose value is the compact JSON of the JSON-Schema parameters object
(`json.dumps(parameters)` in the card, `parameters | tojson` in vLLM's template). The
`IMPORTANT:` instruction block is appended once, after the last tool.
## Tool-call format
The model emits one batch wrapper containing one or more calls. Each call is
`name <|tool▁sep|> arguments`, where **arguments is a raw JSON object string** (no code
fence). Minimal single call (what the model generates after the `<|Assistant|></think>`
prefix):
```text
<|tool▁calls▁begin|><|tool▁call▁begin|>get_weather<|tool▁sep|>{"location": "San Francisco, CA"}<|tool▁call▁end|><|tool▁calls▁end|>
```
Grammar (V3.1):
```text
<|tool▁calls▁begin|><|tool▁call▁begin|>{name}<|tool▁sep|>{json_args}<|tool▁call▁end|>{…more calls…}<|tool▁calls▁end|>
```
- `{name}` must exactly match an advertised tool name. It comes **first**, immediately after
`<|tool▁call▁begin|>`.
- `{json_args}` is valid JSON conforming to the tool's parameter schema, inlined directly.
- The whole assistant turn is then closed by the template/server with
`<|end▁of▁sentence|>`.
(V3.1 has **no** `type` field and **no** ` ```json ` fence around arguments — that is the
older R1/V3-0324 convention; see Version differences.)
## Multiple / parallel tool calls
All calls live inside one `<|tool▁calls▁begin|>…<|tool▁calls▁end|>` wrapper. After the
first `<|tool▁call▁begin|>…<|tool▁call▁end|>`, each additional call is **another
`<|tool▁call▁begin|>…<|tool▁call▁end|>` chained directly, with no separator, newline, or
space between calls** (the card: "chain them directly without separators or spaces"):
```text
<|tool▁calls▁begin|><|tool▁call▁begin|>get_weather<|tool▁sep|>{"location": "San Francisco, CA"}<|tool▁call▁end|><|tool▁call▁begin|>get_weather<|tool▁sep|>{"location": "Seattle, WA"}<|tool▁call▁end|><|tool▁calls▁end|>
```
Note that `<|tool▁calls▁begin|>` (plural, id 128806) appears exactly once; each call uses
the singular `<|tool▁call▁begin|>` (id 128808) / `<|tool▁call▁end|>` (id 128809).
## Tool-result format
Executed results are fed back as `tool`-role messages. In **V3.1** each result is wrapped in
the singular output tokens, with **no** plural `<|tool▁outputs▁…|>` wrapper, emitted right
after the assistant tool-call turn's `<|end▁of▁sentence|>`:
```text
<|tool▁output▁begin|>{result_text}<|tool▁output▁end|>
```
`{result_text}` is the raw tool output (typically a JSON string, but any text). For multiple
results, the V3.1 template emits one `<|tool▁output▁begin|>…<|tool▁output▁end|>` per `tool`
message, concatenated directly. There is **no tool-call ID in the wire format** — results are
matched to calls **positionally** (order of outputs ↔ order of calls).
The model then produces its final answer **directly after `<|tool▁output▁end|>`** with no
`<|Assistant|>` marker and no `</think>` (see Parsing notes — the V3.1 reference template
deliberately renders post-tool assistant content as just `content<|end▁of▁sentence|>`).
> R1-0528 / V3-0324 differ: results are enclosed in a `<|tool▁outputs▁begin|>` …
> `<|tool▁outputs▁end|>` batch wrapper, with each result as
> `<|tool▁output▁begin|>…<|tool▁output▁end|>` and multiple results newline-separated.
## End-to-end example
A complete DeepSeek-V3.1 **non-thinking** multi-turn exchange. Everything is one flat string;
inline `←` comments mark where the model's generation begins (they are not part of the
stream). Whitespace inside the `## Tools` block is literal newlines.
```text
<|begin▁of▁sentence|>You are a helpful assistant.
## Tools
You have access to the following tools:
### get_weather
Description: Get the current weather for a location
Parameters: {"type": "object", "properties": {"location": {"type": "string", "description": "City and state, e.g. San Francisco, CA"}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}}, "required": ["location"]}
IMPORTANT: ALWAYS adhere to this exact format for tool use:
<|tool▁calls▁begin|><|tool▁call▁begin|>tool_call_name<|tool▁sep|>tool_call_arguments<|tool▁call▁end|>{additional_tool_calls}<|tool▁calls▁end|>
Where:
- `tool_call_name` must be an exact match to one of the available tools
- `tool_call_arguments` must be valid JSON that strictly follows the tool's Parameters Schema
- For multiple tool calls, chain them directly without separators or spaces
<|User|>What's the weather in San Francisco?<|Assistant|></think><|tool▁calls▁begin|><|tool▁call▁begin|>get_weather<|tool▁sep|>{"location": "San Francisco, CA", "unit": "celsius"}<|tool▁call▁end|><|tool▁calls▁end|><|end▁of▁sentence|><|tool▁output▁begin|>{"temperature": 18, "unit": "celsius", "condition": "Foggy"}<|tool▁output▁end|>It's currently 18°C and foggy in San Francisco.<|end▁of▁sentence|>
```
Reading the spans:
1. `<|begin▁of▁sentence|>` + system text + `\n\n` + `## Tools…` block — prompt prefix.
2. `<|User|>What's the weather in San Francisco?` — user turn.
3. `<|Assistant|></think>` — non-thinking generation prefix (prompt). **Model generates from here.**
4. `<|tool▁calls▁begin|>…<|tool▁calls▁end|>` — the model's tool call; server appends `<|end▁of▁sentence|>` and stops with `finish_reason: "tool_calls"`.
5. `<|tool▁output▁begin|>…<|tool▁output▁end|>` — your executed result, appended to the prompt.
6. `It's currently 18°C and foggy in San Francisco.<|end▁of▁sentence|>` — **the model generates the final answer directly after the tool output** (no new `<|Assistant|>` marker), ending with EOS.
## OpenAI-compatible API mapping
When fronted by an OpenAI-compatible server (e.g. vLLM with `--tool-call-parser
deepseek_v31`):
- **`finish_reason`**: `"tool_calls"` when the model emitted a `<|tool▁calls▁begin|>…`
batch; otherwise `"stop"`.
- **`message.tool_calls[]`**: one element per `<|tool▁call▁begin|>…<|tool▁call▁end|>`.
- `.type` = `"function"`.
- `.function.name` = the text between `<|tool▁call▁begin|>` and `<|tool▁sep|>`.
- `.function.arguments` = the text between `<|tool▁sep|>` and `<|tool▁call▁end|>`, returned
as a **JSON string** (per the OpenAI spec), not a nested object. The model already emits
raw JSON there, so it is passed through.
- `.id` = **synthesized by the server** (e.g. `chatcmpl-tool-…`). DeepSeek's wire format
carries no call ID.
- **Tool result messages**: `{"role": "tool", "tool_call_id": "<id>", "content": "<result>"}`.
The server renders `content` into `<|tool▁output▁begin|>…<|tool▁output▁end|>`. Because the
prompt has no IDs, `tool_call_id` is used only for client-side bookkeeping; **the model
relies on ordering**, so preserve the order of results relative to the calls.
- **Assistant replay**: when you send a prior assistant turn back with `tool_calls`, the
template inlines `function.arguments`. The HF reference template inlines it **verbatim**
(assumes it is already a JSON string); vLLM's `tool_chat_template_deepseekv31.jinja` pipes
it through `| tojson`. Send `arguments` as a JSON **string** per the OpenAI spec (see the
gotcha below about double-encoding).
## Parsing notes & gotchas
- **Unicode is load-bearing.** Match `|` = U+FF5C and `▁` = U+2581 exactly. ASCII
`<|tool_calls_begin|>` will not tokenize to the special tokens. `<think>`/`</think>` use
ASCII brackets; the rare `<|EOT|>` uses ASCII pipes.
- **Tool/role markers are `special: false`.** Only `<|begin▁of▁sentence|>`,
`<|end▁of▁sentence|>`, `<|▁pad▁|>`, and `<|EOT|>` are flagged `special: true`. So
decoding with `skip_special_tokens=True` will **not** strip `<|tool▁calls▁begin|>`,
`<|tool▁sep|>`, `<|Assistant|>`, `</think>`, etc. — they remain in the decoded string for
the parser to find. (Conversely, do not assume special-token filtering removes them.)
- **No code fence / no `type` field in V3.1.** A parser written for R1/V3-0324
(`function<|tool▁sep|>name` + ` ```json ` block) will not parse V3.1, and vice-versa.
V3.1 is `name<|tool▁sep|>raw_json`.
- **Chaining has no delimiter in V3.1.** Calls abut directly:
`…<|tool▁call▁end|><|tool▁call▁begin|>…`. Do not split on newlines/whitespace; split on
the `<|tool▁call▁begin|>` / `<|tool▁call▁end|>` boundaries. (R1/V3-0324 put a `\n` before
each subsequent call.)
- **No tool-call IDs on the wire.** Match results to calls by position. A server must
generate synthetic `tool_call_id`s for the OpenAI shape.
- **`</think>` appears even in non-thinking mode.** Strip the leading `</think>` (and any
preceding reasoning) before treating the remainder as the visible answer; the template does
`content.split('</think>', 1)[1]` when replaying stored turns.
- **Post-tool generation prompt quirk.** The reference V3.1 chat template only appends the
`<|Assistant|></think>` generation prefix when the **last message is `user`**. After a
`tool` message it appends nothing and the model continues straight after
`<|tool▁output▁end|>`. Agent loops that re-template a conversation ending in a tool result
must not expect (or double-insert) an assistant marker there.
- **`arguments` double-encoding risk.** On replay, vLLM's example template applies
`arguments | tojson`. If `arguments` is already a JSON string (the OpenAI convention), that
pipe will JSON-encode the string again (wrapping it in quotes and escaping it). Pass an
object where the template expects `| tojson`, or a string where the template inlines
verbatim — match the template you actually run.
- **Streaming.** Tool calls arrive token-by-token; the name is complete only at
`<|tool▁sep|>`, and arguments are partial JSON until `<|tool▁call▁end|>`. Buffer per call
boundary; do not attempt to `json.loads` arguments before the closing tool-call token.
- **Malformed output.** With `tool_choice="auto"` and no structural-tag constraint
(`VLLM_ENFORCE_STRICT_TOOL_CALLING=false`), the model can emit invalid JSON in
`tool_call_arguments` or a `tool_call_name` that does not match any tool; the parser
extracts best-effort. Named/`required` tool choice uses the structured-outputs backend and
guarantees schema-valid arguments.
## Version differences: V3.1 vs V3-0324 / R1-0528
The pre-V3.1 models (DeepSeek-V3-0324 and DeepSeek-R1-0528) share an older tool-call
encoding, served in vLLM with `--tool-call-parser deepseek_v3`. The per-call body is:
````text
<|tool▁call▁begin|>function<|tool▁sep|>{name}
```json
{json_args}
```<|tool▁call▁end|>
````
Differences from V3.1:
| Aspect | V3.1 (`deepseek_v31`) | V3-0324 / R1-0528 (`deepseek_v3`) |
| --- | --- | --- |
| Field order in a call | `{name}<|tool▁sep|>{args}` | `function<|tool▁sep|>{name}` (the literal `type`, then name) |
| Arguments wrapping | raw JSON, inline | fenced ` ```json … ``` ` block (name and args separated by `\n`) |
| Chaining of calls | abut directly, **no separator** | each subsequent call prefixed with `\n` |
| Tool results | `<|tool▁output▁begin|>…<|tool▁output▁end|>` per message, no batch wrapper | wrapped in `<|tool▁outputs▁begin|>…<|tool▁outputs▁end|>`, results newline-separated |
| User→assistant boundary | user turn = `<|User|>{q}`; `<|Assistant|></think>` added at generation | user turn = `<|User|>{q}<|Assistant|>` (assistant marker appended in the user branch) |
| Thinking | hybrid; `thinking` kwarg toggles `<think>` vs `</think>` prefix | R1-0528 always reasoning (bare `<|Assistant|>` generation prefix, model opens `<think>` itself); V3-0324 non-reasoning |
| vLLM parser | `--tool-call-parser deepseek_v31` | `--tool-call-parser deepseek_v3` |
Example R1-0528 / V3-0324 parallel call with its result batch:
````text
<|tool▁calls▁begin|><|tool▁call▁begin|>function<|tool▁sep|>get_weather
```json
{"location": "San Francisco, CA"}
```<|tool▁call▁end|>
<|tool▁call▁begin|>function<|tool▁sep|>get_weather
```json
{"location": "Seattle, WA"}
```<|tool▁call▁end|><|tool▁calls▁end|><|end▁of▁sentence|><|tool▁outputs▁begin|><|tool▁output▁begin|>{"temperature": 18}<|tool▁output▁end|>
<|tool▁output▁begin|>{"temperature": 14}<|tool▁output▁end|><|tool▁outputs▁end|>
````
The `deepseek_r1` **reasoning** parser (`--reasoning-parser deepseek_r1`) applies to the R1
series **and** to DeepSeek-V3.1; it extracts the `<think>…</think>` span into the response's
`reasoning` field. It is independent of the tool-call parser.
## Sources
- DeepSeek-V3.1 model card (Chat Template / ToolCall sections): <https://huggingface.co/deepseek-ai/DeepSeek-V3.1>
- DeepSeek-V3.1 `assets/chat_template.jinja`: <https://huggingface.co/deepseek-ai/DeepSeek-V3.1/resolve/main/assets/chat_template.jinja>
- DeepSeek-V3.1 `tokenizer_config.json` (`chat_template`, byte-identical to the jinja): <https://huggingface.co/deepseek-ai/DeepSeek-V3.1/resolve/main/tokenizer_config.json>
- DeepSeek-V3.1 `tokenizer.json` (`added_tokens` → token IDs and `special` flags): <https://huggingface.co/deepseek-ai/DeepSeek-V3.1/resolve/main/tokenizer.json>
- DeepSeek-V3.1 `config.json` (`bos_token_id`, `eos_token_id`, `vocab_size`): <https://huggingface.co/deepseek-ai/DeepSeek-V3.1/resolve/main/config.json>
- DeepSeek-R1-0528 model card and `tokenizer_config.json` (older tool format): <https://huggingface.co/deepseek-ai/DeepSeek-R1-0528> · <https://huggingface.co/deepseek-ai/DeepSeek-R1-0528/resolve/main/tokenizer_config.json>
- DeepSeek-R1 model card: <https://huggingface.co/deepseek-ai/DeepSeek-R1>
- DeepSeek-V3-0324 `tokenizer_config.json` (older tool format): <https://huggingface.co/deepseek-ai/DeepSeek-V3-0324/resolve/main/tokenizer_config.json>
- vLLM tool-call template for V3.1 (`## Tools` injection + `| tojson`): <https://github.com/vllm-project/vllm/blob/main/examples/tool_chat_template_deepseekv31.jinja>
- vLLM Tool Calling docs (`deepseek_v3`, `deepseek_v31` parser flags): <https://docs.vllm.ai/en/latest/features/tool_calling/>
- vLLM Reasoning Outputs docs (`deepseek_r1` reasoning parser; V3.1 thinking default): <https://docs.vllm.ai/en/latest/features/reasoning_outputs/>
+145
View File
@@ -0,0 +1,145 @@
# Gemini Pythonic tool-calling format (`tool_code` / `default_api`)
Tool-calling convention of Google's hosted **Gemini** models (current generation, incl. `gemini-3.5-flash` / `*-pro` / `*-preview`) and the **Gemma 3** open-weights family. Both drive tool use **entirely through prompt engineering** — there are **no dedicated special tokens**. The model emits each invocation as **Python source**: a call `default_api.<function_name>(<kwargs>)`, conventionally wrapped in `print(...)` and placed inside a fenced ```` ```tool_code ```` block; it reads results back from a ```` ```tool_outputs ```` block. Because the mechanism is plain text the model was post-trained to produce, the exact same syntax periodically leaks into ordinary output (surfaced by Vertex/AI-Studio as `finish_reason = MALFORMED_FUNCTION_CALL`) — that leak is the clearest public evidence of the format.
Verified against: the official Gemma 3 function-calling guide (`ai.google.dev/gemma/docs/capabilities/function-calling` — the two recommended prompts, one Pythonic and one JSON), Simon Willison's transcription of those two prompts, Philipp Schmid's Gemma 3 walkthrough (`philschmid.de/gemma-function-calling`), and the reverse-engineered hosted-Gemini form recovered from `MALFORMED_FUNCTION_CALL` reports: `google/adk-go#492` (`Malformed function call: print(default_api.`), `google-gemini/cookbook#929` (`executableCode` part = `print(default_api.get_complaint_number_tool(consumer_number_or_mobile_number='2001234567'))`), `firebase/genkit#2628` (the ```` ```tool_code ```` markdown wrapper), and the Google AI dev-forum thread "Gemini 2 flash returns raw markdown instead of function call" (71964).
## "Special" tokens
**None.** Nothing here is a control token in the tokenizer's special-token table — every marker below BPE-splits into ordinary text and survives a `skip_special_tokens=True` decode. This is the defining property of the convention and the reason it both (a) works across hosted Gemini and open Gemma without tokenizer support and (b) leaks. The functional markers are:
| Marker (verbatim) | Role |
|---|---|
| ` ```tool_code ` | Opens a fenced block whose body is Python the app must execute. Closed by a bare ` ``` `. |
| ` ```tool_outputs ` | Opens a fenced block carrying the executed results back to the model. Closed by a bare ` ``` `. |
| `default_api` | Synthetic module namespace the hosted stack bundles un-namespaced tools into. Calls read `default_api.<name>(...)`. |
| `print(...)` | Conventional wrapper around the call in the hosted-Gemini form (the model is trained to "print" the call). Semantically irrelevant — the runtime parses the call, it does not execute Python. |
There is **no** per-call id on the wire and **no** in-band reasoning marker — Gemini reasoning travels out of band as API "thought signatures", never as `<think>`-style text.
## Roles / turn structure
The Pythonic payload is independent of the envelope, and the envelope differs by deployment:
- **Hosted Gemini** uses the normal `contents[]` turn structure (`role: "user" | "model"`); the `tool_code` block appears inside a `model` turn's text, and `tool_outputs` is supplied as the next turn.
- **Gemma 3** (open weights) uses the Gemma chat template (`<start_of_turn>user … <end_of_turn>` / `<start_of_turn>model`); the tool prompt is prepended to the first user turn and the blocks live inside model/user turns.
This document specifies the **payload** (the two fenced blocks + the Python call form); the surrounding turn tokens belong to whichever template hosts it.
## Tool definitions
Tools are advertised in the prompt as a JSON-Schema catalog. Gemma 3's official guide ships **two** interchangeable system-prompt templates that differ only in how the model is told to answer:
1. **Pythonic** (the one this spec targets):
> You have access to functions. If you decide to invoke any of the function(s), you MUST put it in the format of `[func_name1(params_name1=params_value1, params_name2=params_value2...), func_name2(params)]`
> You SHOULD NOT include any other text in the response if you call a function
2. **JSON** (the sibling convention — see `qwen3.md` for the closely related Hermes shape):
> … you MUST put it in the format of `{"name": function name, "parameters": dictionary of argument name and its value}`
Hosted Gemini wraps the same idea in markdown fences and the `default_api` namespace. The function signatures themselves are passed as OpenAI-style tool JSON (`{"type":"function","function":{name,description,parameters}}`).
## Tool-call format
One call is a Python call expression. The hosted-Gemini canonical form is a `print()` of a `default_api` method, fenced:
````text
```tool_code
print(default_api.get_current_temperature(location="London", unit="celsius"))
```
````
All of the following are accepted equivalents seen in the wild and across Gemma/Gemini variants; a robust parser normalizes them to `{name, arguments}`:
- `print(default_api.NAME(KWARGS))` — hosted Gemini canonical.
- `default_api.NAME(KWARGS)` — `print`/namespace are optional sugar.
- `NAME(KWARGS)` — bare call (Gemma 3 Pythonic prompt).
- `result = NAME(KWARGS)` — assignment form (Gemma 3 docs use `result = convert(...)`).
Argument values are **Python literals**, not JSON:
| Python literal | Example | Decoded |
|---|---|---|
| string | `'London'` or `"London"` | `"London"` |
| int / float | `42`, `3.14` | `42`, `3.14` |
| bool | `True` / `False` | `true` / `false` |
| null | `None` | `null` |
| list | `["a", "b"]` | `["a","b"]` |
| dict | `{"k": 1}` | `{"k":1}` |
Strings use Python escaping (`\n`, `\t`, `\\`, `\'`, `\"`); hosted Gemini emits single quotes (`location='London'`), Gemma examples use double quotes — both are valid. Arguments are keyword form (`name=value`); positional arguments are not used because the runtime maps to a named schema.
## Multiple / parallel tool calls
Two encodings exist, both inside a single `tool_code` block:
- **Gemma 3 Pythonic prompt** — a Python **list** of call expressions:
````text
```tool_code
[get_current_temperature(location="London"), get_temperature_date(location="London", date="2024-10-01")]
```
````
- **Hosted Gemini** — one `print(default_api...)` **statement per line**:
````text
```tool_code
print(default_api.get_current_temperature(location="London"))
print(default_api.get_temperature_date(location="London", date="2024-10-01"))
```
````
Either way the calls are returned in source order; the application executes them and returns one result per call in the same order.
## Tool-result format
Executed results are returned to the model in a ```` ```tool_outputs ```` block. Gemma 3 docs use assignment-style values (`result = 92.3`); for opaque tool output the block simply carries the returned text/JSON:
````text
```tool_outputs
{"temperature": 26.1, "location": "London", "unit": "celsius"}
```
````
The model then continues with either a natural-language answer or another `tool_code` block.
## End-to-end example
````text
<user>
What's the temperature in London?
<model>
```tool_code
print(default_api.get_current_temperature(location="London", unit="celsius"))
```
<user>
```tool_outputs
{"temperature": 11.4, "location": "London", "unit": "celsius"}
```
<model>
It's currently 11.4°C in London.
````
## OpenAI-compatible / native API mapping
- Hosted Gemini's native API normally returns a structured `functionCall` part (`{name, args}`); for Gemini 3 each carries an `id` that must be echoed in the matching `functionResponse`, plus a `thoughtSignature` that must be preserved. The Pythonic text form is what you get when the structured path *fails* (`finish_reason = MALFORMED_FUNCTION_CALL`) or when tool use is driven purely by prompt (Gemma, or Gemini via the code-execution `executableCode` part).
- When parsed out of an OpenAI-compatible shim, each recovered call becomes `tool_calls[i] = {id (server-minted), type:"function", function:{name, arguments:<JSON string>}}` — the Python kwargs are re-serialized to a JSON string at that boundary.
- Feed results back as the deployment's tool/`functionResponse` turn (hosted) or a `tool_outputs` block in the next user turn (prompt-driven).
## Parsing notes & gotchas
- **Python, not JSON.** `True`/`False`/`None` (not `true`/`false`/`null`), single-quoted strings, and trailing commas are all legal. A JSON parser will reject valid calls; decode Python literals.
- **Strip the wrapper.** Normalize away `print(...)`, a `default_api.` (or any `module.`) prefix, and an `LHS =` assignment before reading the call name. `print` is never a tool name.
- **Skip string contents when scanning.** A call like `search(pattern="foo(")` contains a `(` inside a string; a naive `\w+\(` scan mis-detects `foo` as a callee. Track string state and only treat top-level `(` as a call opener.
- **Fence ambiguity.** The body terminates at the first bare ` ``` `; a string argument literally containing ` ``` ` will truncate the block early (rare, accepted limitation).
- **It leaks.** Because nothing is a special token, the format appears verbatim in normal responses when the model "decides" to call a tool but the structured decoder misfires. Production code reading raw text should detect ` ```tool_code ` and parse it; production code on the structured API should retry on `MALFORMED_FUNCTION_CALL`.
- **Variant divergence.** Gemma **4** abandoned this Pythonic form for a token-delimited brace syntax (`<|tool_call>call:NAME{…}<tool_call|>`) — a different convention documented in `gemma.md`. This spec covers hosted Gemini and Gemma 3.
## Sources
- Gemma 3 function calling (two recommended prompts): https://ai.google.dev/gemma/docs/capabilities/function-calling
- Simon Willison, "Function calling with Gemma": https://simonwillison.net/2025/Mar/26/function-calling-with-gemma/
- Philipp Schmid, "Google Gemma 3 Function Calling Example": https://www.philschmid.de/gemma-function-calling
- Gemini 3 thought signatures + functionCall ids: https://ai.google.dev/gemini-api/docs/gemini-3
- `default_api` / `tool_code` leak evidence: https://github.com/google/adk-go/issues/492 · https://github.com/google-gemini/cookbook/issues/929 · https://github.com/firebase/genkit/issues/2628 · https://discuss.ai.google.dev/t/gemini-2-flash-api-returns-raw-markdown-instead-of-function-call/71964
+104
View File
@@ -0,0 +1,104 @@
# Gemma 4 tool-calling format (token-delimited `call:NAME{…}`)
Tool-calling convention of Google's **Gemma 4** open-weights family (`google/gemma-4-*-it`). It is a clean break from the prompt-engineered Pythonic `tool_code` form used by Gemma 3 and hosted Gemini (see `gemini.md`): Gemma 4 introduces **dedicated special tokens** and a compact **token-delimited brace syntax**. Tool declarations, calls, and responses each get their own paired markers, and every string value is wrapped in a `<|"|>` token rather than ASCII quotes. The model emits one call as `<|tool_call>call:NAME{key:value,…}<tool_call|>`; the developer parses it, runs the tool, and appends `<|tool_response>response:NAME{…}<tool_response|>`.
Verified against: the official "Function calling with Gemma 4" guide (`ai.google.dev/gemma/docs/capabilities/text/function-calling-gemma4`), including the byte-exact `processor.apply_chat_template(...)` renderings and the reference `extract_tool_calls` regex it ships. All example streams below are copied from that page (model `google/gemma-4-E2B-it`).
## Special tokens
Gemma 4 wraps each structural element in a paired token. Note the **asymmetric pipe placement** — an opener carries the pipe on the left (`<|x>`) and its closer carries it on the right (`<x|>`):
| Open | Close | Purpose |
|---|---|---|
| `<bos>` | — | Beginning of sequence |
| `<|turn>` | `<turn|>` | One conversation turn; the role name is the first line of the body |
| `<|tool>` | `<tool|>` | A tool **declaration** block (in the system turn) |
| `<|tool_call>` | `<tool_call|>` | One tool **call** emitted by the model |
| `<|tool_response>` | `<tool_response|>` | One tool **result** fed back to the model |
| `<|"|>` | `<|"|>` | String-literal delimiter (same token on both ends) |
| `<eos>` | — | End of sequence |
Because the string delimiter is a token (`<|"|>`), values may contain raw ASCII quotes and commas without escaping — only a literal `<|"|>` token sequence cannot appear inside a string.
## Roles / turn structure
Each turn is `<|turn>{role}\n{body}<turn|>`. Roles are `system`, `user`, `model`. With a generation prompt the stream ends at `<|turn>model\n` and the model continues. Tool declarations are merged into the `system` turn; tool calls and the following tool responses are emitted inside the `model` turn (the response block immediately follows the call block in the re-rendered history).
## Tool definitions
Each tool is declared in the system turn as `<|tool>declaration:NAME{…}<tool|>`, where the body is the schema serialized in the same brace syntax used by calls. Types are upper-cased strings (`STRING`, `OBJECT`, …). Byte-exact, from the guide:
```text
<|tool>declaration:get_current_temperature{description:<|"|>Gets the current temperature for a given location.<|"|>,parameters:{properties:{location:{description:<|"|>The city name, e.g. San Francisco<|"|>,type:<|"|>STRING<|"|>} },required:[<|"|>location<|"|>],type:<|"|>OBJECT<|"|>} }<tool|>
```
## Tool-call format
The model emits one call per `<|tool_call>…<tool_call|>` block. The body is `call:NAME{ARGS}`, where `ARGS` is a comma-separated list of `key:value` pairs:
```text
<|tool_call>call:get_current_temperature{location:<|"|>London<|"|>}<tool_call|>
```
Value grammar inside `{…}`:
| Value kind | Encoding | Example |
|---|---|---|
| string | `<|"|>text<|"|>` | `location:<|"|>London<|"|>` |
| int / float | bare | `count:42` |
| bool | bare | `flag:true` |
| null | bare | `unit:null` |
| list | `[v,v,…]` | `tags:[<|"|>a<|"|>,<|"|>b<|"|>]` |
| nested object | `{k:v,…}` | `config:{theme:<|"|>dark<|"|>}` |
The reference parser shipped in the guide:
```python
[{
"name": name,
"arguments": {
k: cast((v1 or v2).strip())
for k, v1, v2 in re.findall(r'(\w+):(?:<\|"\|>(.*?)<\|"\|>|([^,}]*))', args)
}
} for name, args in re.findall(r"<\|tool_call>call:(\w+)\{(.*?)\}<tool_call\|>", text, re.DOTALL)]
```
i.e. each argument value is either a `<|"|>…<|"|>` string or a bare run of non-`,}` characters (cast to int/float/bool, else kept as a string).
## Multiple / parallel tool calls
Parallel calls are consecutive `<|tool_call>…<tool_call|>` blocks (one call each), returned in order. The application returns one `<|tool_response>` per call in the same order.
## Tool-result format
Each result is `<|tool_response>response:NAME{…}<tool_response|>`, the response object serialized in the same brace syntax. Byte-exact, from the guide's re-rendered history:
```text
<|tool_response>response:get_current_weather{temperature:15,weather:<|"|>sunny<|"|>}<tool_response|>
```
## End-to-end example
Byte-exact `apply_chat_template` output from the guide (system + tool, user, model call, tool response, final answer — note the response block sits in the same model turn, right after the call):
```text
<bos><|turn>system
You are a helpful assistant.<|tool>declaration:get_current_weather{description:<|"|>Gets the current weather in a given location.<|"|>,parameters:{properties:{location:{description:<|"|>The city and state, e.g. "San Francisco, CA" or "Tokyo, JP"<|"|>,type:<|"|>STRING<|"|>},unit:{description:<|"|>The unit to return the temperature in.<|"|>,enum:[<|"|>celsius<|"|>,<|"|>fahrenheit<|"|>],type:<|"|>STRING<|"|>} },required:[<|"|>location<|"|>],type:<|"|>OBJECT<|"|>} }<tool|><turn|>
<|turn>user
Hey, what's the weather in Tokyo right now?<turn|>
<|turn>model
<|tool_call>call:get_current_weather{location:<|"|>Tokyo, JP<|"|>}<tool_call|><|tool_response>response:get_current_weather{temperature:15,weather:<|"|>sunny<|"|>}<tool_response|>The current weather in Tokyo is 15 degrees Celsius and sunny.<turn|>
```
## Parsing notes & gotchas
- **String delimiter is a token, not a quote.** Inside `<|"|>…<|"|>` the bytes `"` and `,` are literal data — the example `<|"|>The city and state, e.g. "San Francisco, CA"…<|"|>` contains both. Split arguments on `,`/`}` only **outside** a `<|"|>…<|"|>` span.
- **Asymmetric pipes.** The closer is `<tool_call|>`, not `</tool_call>` or `<|tool_call>`. Matching the wrong pipe side will never close the block.
- **One call per block.** Unlike a JSON `tool_calls[]` array, parallelism is "more blocks", not "more entries in one block".
- **Bare scalars.** A value not wrapped in `<|"|>` is `true`/`false` → bool, `null`/`none` → null, numeric → number, otherwise a bare string (e.g. an unquoted enum or type name like `STRING`).
- **Not Gemma 3 / hosted Gemini.** Those use the Pythonic `tool_code` / `default_api` form in `gemini.md`. Gemma 4 replaced it with this token syntax; the two are not interchangeable.
## Sources
- Function calling with Gemma 4 (byte-exact chat-template renderings + reference parser): https://ai.google.dev/gemma/docs/capabilities/text/function-calling-gemma4
- Gemma 4 prompt formatting: https://ai.google.dev/gemma/docs/core/prompt-formatting-gemma4
+296
View File
@@ -0,0 +1,296 @@
# GLM-4.5 / GLM-4.6 tool-calling format
Native tool-calling convention of Zhipu AI / Z.ai's **GLM-4.5** family (`zai-org/GLM-4.5` 355B-A32B and `zai-org/GLM-4.5-Air` 106B-A12B, `model_type: "glm4_moe"`), shared byte-for-byte by **GLM-4.6**. Unlike the JSON-in-a-tag conventions used by most families, GLM emits each tool call as an **XML-like** block: `<tool_call>{name}` followed by alternating `<arg_key>`/`<arg_value>` element pairs, closed by `</tool_call>`. The prompt is a GLM-style sequence opened by `[gMASK]<sop>` with turn markers `<|system|>`, `<|user|>`, `<|assistant|>`, `<|observation|>`. An inference server turns the raw stream into OpenAI-style `tool_calls` with a parser plus a reasoning parser: both vLLM and SGLang expose `--tool-call-parser glm45 --reasoning-parser glm45` (vLLM additionally needs `--enable-auto-tool-choice`). Tool calling and reasoning are driven entirely by the bundled `chat_template.jinja`; thinking mode is on by default and is disabled per-request with `chat_template_kwargs={"enable_thinking": false}`.
This document was verified against the authoritative `chat_template.jinja` from the HF repo (fetched raw and **rendered locally with Jinja2** — `trim_blocks=True, lstrip_blocks=True`, transformers' `tojson` filter — to produce the byte-exact streams below), `tokenizer_config.json` and `generation_config.json` for the exact token IDs and stop tokens, the model card, and the vLLM (`Glm4MoeModelToolParser`) and SGLang (`Glm4MoeDetector`) parser sources. The HF `resolve`/`blob` web paths redirect to the model-card API; the byte-exact source was obtained via the `resolve/main/...:raw` cache (template commit `cbb2c7cfb52fa128a9660cb1a7a78e017899e115`). The GLM-4.5 and GLM-4.6 `chat_template.jinja` files are identical (same content hash `41478957…`).
## Special tokens
Token IDs are from `tokenizer_config.json` (`added_tokens_decoder`). Note the split: the turn/role markers are registered as **special** tokens, whereas the structural tool-call and thinking tags are each a single dedicated vocabulary token but flagged **`special: false`** (they are emitted/printed as ordinary text, not stripped as control tokens).
| Token (verbatim) | ID | `special` | Purpose |
|---|---|---|---|
| `[gMASK]` | 151331 | true | GLM prefix / blank-infilling sentinel; first token of every prompt |
| `<sop>` | 151333 | true | "Start of piece" — immediately follows `[gMASK]` to open the sequence |
| `<eop>` | 151334 | true | "End of piece" (not emitted by the chat template) |
| `<\|system\|>` | 151335 | true | Opens a system turn (and the injected tools turn) |
| `<\|user\|>` | 151336 | true | Opens a user turn (also an EOS id — see below) |
| `<\|assistant\|>` | 151337 | true | Opens an assistant turn / generation prompt |
| `<\|observation\|>` | 151338 | true | Opens a tool-result (observation) turn (also an EOS id) |
| `<\|endoftext\|>` | 151329 | true | End-of-text; `eos_token` and `pad_token` |
| `<think>` | 151350 | false | Opens the reasoning span inside an assistant turn |
| `</think>` | 151351 | false | Closes the reasoning span |
| `<tool_call>` | 151352 | false | Opens one tool call; function name follows on the same line |
| `</tool_call>` | 151353 | false | Closes one tool call |
| `<arg_key>` | 151356 | false | Opens an argument-name element |
| `</arg_key>` | 151357 | false | Closes an argument-name element |
| `<arg_value>` | 151358 | false | Opens an argument-value element |
| `</arg_value>` | 151359 | false | Closes an argument-value element |
| `<tool_response>` | 151354 | false | Wraps one tool result inside an observation turn |
| `</tool_response>` | 151355 | false | Closes a tool result |
| `/nothink` | 151360 | true | Soft switch appended to user text to suppress thinking |
Notes on exactness:
- All pipes are ASCII `|` (U+007C); GLM uses no fullwidth `|` (U+FF5C) or `▁` (U+2581) variants (unlike DeepSeek). Reproduce `<|system|>`, `<|user|>`, `<|assistant|>`, `<|observation|>` exactly, and `[gMASK]` with literal square brackets.
- Because `<tool_call>`, `<arg_key>`, `<arg_value>`, `<tool_response>`, `<think>` (and their closers) each map to exactly **one** token ID, they cost one token apiece in the stream — but being `special: false` they round-trip through detokenization as plain text. Parsers therefore match them as literal substrings in the decoded text, not as control-token ids.
- `eos_token_id` is a **list**: `[151329, 151336, 151338]` = `<|endoftext|>`, `<|user|>`, `<|observation|>` (from `generation_config.json`). This is how a tool-call turn ends: after `</tool_call>` the model emits `<|observation|>`, which is an EOS id, so generation halts and the server reports a tool call (see Turn structure).
## Roles / channels / turn structure
Every prompt begins with the literal two-token prefix `[gMASK]<sop>` (no following newline). Turns are then concatenated, each introduced by its role marker; there is no per-turn terminator token in rendered history (the next marker, or an EOS id during generation, ends a turn).
- **System** (`<|system|>`): role marker, newline, then the message text. When `tools` are supplied, a synthetic tools system turn is rendered **first**, before any user-supplied system turn (the two are separate `<|system|>` blocks — see Tool definitions).
- **User** (`<|user|>`): role marker, newline, then text. If `enable_thinking` is false, the literal `/nothink` is appended to the user text (unless it already ends with `/nothink`).
- **Assistant** (`<|assistant|>`): role marker, then a reasoning span and/or visible content and/or tool calls. The reasoning span is `\n<think>{reasoning}</think>`; visible content follows on its own line; tool calls follow as `<tool_call>…</tool_call>` blocks.
- **Tool result** (`<|observation|>`): role marker introducing one or more `<tool_response>…</tool_response>` blocks (see Tool-result format).
Thinking / reasoning channel:
- Reasoning lives in `<think>…</think>` inside the assistant turn. The `--reasoning-parser glm45` extracts it into a separate `reasoning_content` field; the visible answer is whatever follows `</think>`.
- **Only the reasoning of assistant turns after the last user message is kept.** The template renders every earlier assistant turn with an empty `<think></think>` and drops its `reasoning_content` (or any inline `<think>…</think>` embedded in `content`). This keeps stale chains of thought out of the context on later turns.
- An assistant turn with neither preserved reasoning nor an explicit chain renders `\n<think></think>` (empty), then content/tool calls.
Generation prompt (`add_generation_prompt=True`):
- **Thinking mode (default):** the prompt ends with a bare `<|assistant|>`; the model continues with `\n<think>…</think>` then its answer or tool calls.
- **Non-thinking mode** (`enable_thinking=false`): the prompt ends with `<|assistant|>\n<think></think>`, pre-filling an empty reasoning span so the model goes straight to the answer.
How a tool-call turn terminates: there is no dedicated "stop after tool call" token. The model emits `</tool_call>` and then `<|observation|>` (token 151338), which is one of the three EOS ids, so decoding stops. The server inspects the text, finds `<tool_call>`, and returns `finish_reason: "tool_calls"`.
## Tool definitions
When the request carries `tools`, the template prepends one `<|system|>` turn containing a fixed preamble, the tool list wrapped in `<tools>…</tools>`, and a literal description of the output format. Each tool is serialized with `tool | tojson(ensure_ascii=False)` — i.e. the **entire OpenAI tool object verbatim**, including the `{"type": "function", "function": {…}}` wrapper, with default JSON spacing (`", "` / `": "`). One tool per line.
```text
<|system|>
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{"type": "function", "function": {"name": "get_weather", "description": "Get current weather for a city", "parameters": {"type": "object", "properties": {"location": {"type": "string", "description": "City name"}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}}, "required": ["location"]}}}
</tools>
For each function call, output the function name and arguments within the following XML format:
<tool_call>{function-name}
<arg_key>{arg-key-1}</arg_key>
<arg_value>{arg-value-1}</arg_value>
<arg_key>{arg-key-2}</arg_key>
<arg_value>{arg-value-2}</arg_value>
...
</tool_call>
```
The `<tool_call>{function-name}` / `<arg_key>` / `<arg_value>` lines above are part of the **prompt text** (the format spec the model is told to follow), not an example call. This tools turn is emitted only when `tools` is non-empty, and it is closed implicitly by the next role marker (e.g. a user-supplied `<|system|>` or the first `<|user|>`), with no blank line between them.
## Tool-call format
The model emits a call as an `<tool_call>` block: the function **name on the same line** as the opening tag, a newline, then one `<arg_key>…</arg_key>` + `<arg_value>…</arg_value>` pair per argument, closed by `</tool_call>`. Minimal single call (assistant generation in thinking mode; reasoning shown for realism):
```text
<think>The user wants the weather in Beijing. I'll call get_weather.</think>
<tool_call>get_weather
<arg_key>location</arg_key>
<arg_value>Beijing</arg_value>
<arg_key>unit</arg_key>
<arg_value>celsius</arg_value>
</tool_call>
```
Anatomy and value encoding (this is the single most error-prone part):
- The function name is the text between `<tool_call>` and the first newline — there is **no** wrapping tag around it and **no** space after `<tool_call>`.
- Each argument is two adjacent elements: `<arg_key>name</arg_key>` then `<arg_value>value</arg_value>`, conventionally one pair per line.
- **Argument values are NOT uniformly JSON.** The template renders each value as `value | tojson(ensure_ascii=False) if value is not string else value`:
- **string** values are emitted **raw, without surrounding quotes** → `<arg_value>Beijing</arg_value>` (not `"Beijing"`).
- **non-string** values (number, boolean, null, object, array) are JSON-encoded → `<arg_value>3</arg_value>`, `<arg_value>true</arg_value>`, `<arg_value>{"k": 1}</arg_value>`.
- A **zero-argument** call has no pairs: the name is followed by a newline and the closer — `<tool_call>get_time\n</tool_call>`.
Because string values lose their quotes, a parser must decide per argument whether to JSON-decode or treat the value as a literal string. Both reference parsers do this by consulting the tool's JSON Schema: if the parameter's type is `string`, the raw text is taken as-is; otherwise the value is JSON-decoded (with `ast.literal_eval` and raw-string fallbacks). The model is trained to follow the schema, so it emits a bare string exactly when the parameter is string-typed.
## Multiple / parallel tool calls
Two or more calls in one turn are emitted as consecutive `<tool_call>…</tool_call>` blocks separated by a single newline (no wrapper element around the set). Raw assistant emission for two parallel calls with mixed argument types:
```text
<think>Two cities. Call get_weather twice in parallel.</think>
<tool_call>get_weather
<arg_key>location</arg_key>
<arg_value>Beijing</arg_value>
<arg_key>unit</arg_key>
<arg_value>celsius</arg_value>
</tool_call>
<tool_call>get_weather
<arg_key>location</arg_key>
<arg_value>Shanghai</arg_value>
<arg_key>days</arg_key>
<arg_value>3</arg_value>
<arg_key>verbose</arg_key>
<arg_value>true</arg_value>
</tool_call>
```
Note `Beijing`/`Shanghai`/`celsius` (string) are bare, while `3` (number) and `true` (boolean) are JSON literals. Parsers split on the non-greedy `<tool_call>.*?</tool_call>` regex, so any number of calls is supported; each becomes a separate entry in `tool_calls[]`.
## Tool-result format
Results are returned in an **observation** turn. For a single result: the `<|observation|>` marker, a newline, then the result wrapped in `<tool_response>` / `</tool_response>`:
```text
<|observation|>
<tool_response>
{"temperature": 26, "unit": "celsius", "condition": "Sunny"}
</tool_response>
```
The content between the tags is inserted **verbatim** (callers typically pass a JSON string, but any text is allowed). For **multiple** results from a set of parallel calls, the `<|observation|>` marker appears **once** and each result gets its own `<tool_response>` block (consecutive `tool`-role messages are merged under a single observation turn):
```text
<|observation|>
<tool_response>
{"temperature": 26, "condition": "Sunny"}
</tool_response>
<tool_response>
{"temperature": 30, "condition": "Cloudy"}
</tool_response>
```
The chat template reads **only** the tool message's `content` — it does not consult any `tool_call_id`. Results are therefore correlated to calls **positionally / by order**, not by an embedded id (GLM's wire format carries no per-call id; see API mapping).
## End-to-end example
A complete multi-turn weather exchange. These are the exact locally rendered streams; newlines inside a turn are literal and turns are otherwise contiguous (no separators between markers).
**Stage 1 — prompt fed to the model** (`tools` set, one prior system message, `add_generation_prompt=True`, thinking mode):
```text
[gMASK]<sop><|system|>
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{"type": "function", "function": {"name": "get_weather", "description": "Get current weather for a city", "parameters": {"type": "object", "properties": {"location": {"type": "string", "description": "City name"}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}}, "required": ["location"]}}}
</tools>
For each function call, output the function name and arguments within the following XML format:
<tool_call>{function-name}
<arg_key>{arg-key-1}</arg_key>
<arg_value>{arg-value-1}</arg_value>
<arg_key>{arg-key-2}</arg_key>
<arg_value>{arg-value-2}</arg_value>
...
</tool_call><|system|>
You are a helpful assistant.<|user|>
What's the weather in Beijing?<|assistant|>
```
**Assistant generation** (model output; it ends by emitting `<|observation|>`, an EOS id, so decoding stops there; server returns `finish_reason: "tool_calls"`):
```text
<think>The user wants the weather in Beijing. I'll call get_weather.</think>
<tool_call>get_weather
<arg_key>location</arg_key>
<arg_value>Beijing</arg_value>
<arg_key>unit</arg_key>
<arg_value>celsius</arg_value>
</tool_call>
```
**Stage 2 — prompt for the next turn**, after appending the assistant tool-call turn and the tool result, then `add_generation_prompt=True`:
```text
[gMASK]<sop><|system|>
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{"type": "function", "function": {"name": "get_weather", "description": "Get current weather for a city", "parameters": {"type": "object", "properties": {"location": {"type": "string", "description": "City name"}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}}, "required": ["location"]}}}
</tools>
For each function call, output the function name and arguments within the following XML format:
<tool_call>{function-name}
<arg_key>{arg-key-1}</arg_key>
<arg_value>{arg-value-1}</arg_value>
<arg_key>{arg-key-2}</arg_key>
<arg_value>{arg-value-2}</arg_value>
...
</tool_call><|system|>
You are a helpful assistant.<|user|>
What's the weather in Beijing?<|assistant|>
<think>The user wants the weather in Beijing. I'll call get_weather.</think>
<tool_call>get_weather
<arg_key>location</arg_key>
<arg_value>Beijing</arg_value>
<arg_key>unit</arg_key>
<arg_value>celsius</arg_value>
</tool_call><|observation|>
<tool_response>
{"temperature": 26, "unit": "celsius", "condition": "Sunny"}
</tool_response><|assistant|>
```
**Final assistant generation** (natural-language answer, terminated by `<|endoftext|>`; `finish_reason: "stop"`):
```text
<think>Got it, 26C and sunny.</think>
It's 26°C and sunny in Beijing right now.
```
Two subtleties visible above: (1) the reasoning of the assistant tool-call turn is **preserved** in Stage 2 only because it is the segment after the last user message; with another user turn after it, that `<think>…</think>` would be re-rendered empty. (2) The tool-call turn and the observation turn abut directly (`</tool_call><|observation|>`), and the observation abuts the next assistant marker (`</tool_response><|assistant|>`).
For **non-thinking** mode the user text carries the soft switch and the generation prompt pre-fills an empty think span:
```text
<|user|>
Hi there/nothink<|assistant|>
<think></think>
```
## OpenAI-compatible API mapping
With a server parser active (`--tool-call-parser glm45 --reasoning-parser glm45`), the raw stream maps onto Chat Completions as follows:
- `choices[].finish_reason` = `"tool_calls"` when the output contained at least one `<tool_call>` (otherwise `"stop"`).
- `choices[].message.content` = the text **before** the first `<tool_call>` (normalized to `null` if empty/whitespace). The `<think>…</think>` reasoning is removed by the reasoning parser and surfaced separately as `message.reasoning_content`.
- `choices[].message.tool_calls[]` — one entry per `<tool_call>…</tool_call>` block:
- `.id` = a **server-generated** id (e.g. vLLM's `make_tool_call_id()`), **not** present in the model output. GLM emits no per-call id in the stream.
- `.type` = `"function"`.
- `.function.name` = the text after `<tool_call>` up to the first newline.
- `.function.arguments` = a **JSON string** (an object), reconstructed from the `<arg_key>`/`<arg_value>` pairs with per-argument typing from the tool schema. vLLM returns `json.dumps(arg_dct, ensure_ascii=False)`, e.g. `"{\"location\": \"Beijing\", \"unit\": \"celsius\"}"`. Clients `json.loads()` it before use.
- **Request side — tool results** are sent back as `role: "tool"` messages, e.g.:
```json
{"role": "tool", "tool_call_id": "call_abc123", "content": "{\"temperature\": 26, \"unit\": \"celsius\", \"condition\": \"Sunny\"}"}
```
The chat template renders only `content` (inside `<tool_response>`); `tool_call_id` is **ignored by the template** and matters only for the client's own bookkeeping. Order results to match the calls.
- **Request side — assistant tool-call history**: the OpenAI shape carries `function.arguments` as a JSON **string**, but the chat template iterates `arguments.items()` and therefore needs an **object**. vLLM/SGLang parse the string back into a dict before rendering; if you call `tokenizer.apply_chat_template` directly, pass `arguments` as a dict (and optionally `reasoning_content` as a string) or the template will raise.
- Disable thinking via `extra_body={"chat_template_kwargs": {"enable_thinking": false}}` (OpenAI Python client) — this flips the template to the `/nothink` + pre-filled `<think></think>` path.
## Parsing notes & gotchas
- **String values are unquoted; typing needs the schema.** The decisive rule: a `<arg_value>` is a literal string iff the parameter is string-typed in the tool's JSON Schema; otherwise it is JSON. vLLM's `_is_string_type` and SGLang's `get_argument_type` both walk `properties[arg].type` (handling `anyOf`/`oneOf`/`enum`/`allOf`/type-arrays). If the schema is missing/loose, they fall back to "try `json.loads`, then `ast.literal_eval`, then treat as string" — so a bare word like `celsius` survives as a string, while `26` becomes a number. A string value that *looks* like JSON (e.g. a parameter typed `string` whose value is `{"a":1}`) is correctly kept as the literal string only because the schema says `string`.
- **Extraction regexes (GLM-4.5/4.6).** vLLM: calls via `<tool_call>.*?</tool_call>` (DOTALL); name/body via `<tool_call>([^\n]*)\n(.*)</tool_call>`; pairs via `<arg_key>(.*?)</arg_key>\s*<arg_value>(.*?)</arg_value>`. The name regex **requires a newline** after the name — matching the 4.5/4.6 template. SGLang uses an equivalent `(?:\\n|\n)` form so it also tolerates literal escaped `\n`.
- **`</arg_value>` in a value breaks parsing.** Values are captured non-greedily up to the next `</arg_value>`; a value whose text contains `</arg_value>` (or `</tool_call>`) truncates early. There is no escaping mechanism in the wire format.
- **Tool calls are parsed from `content` only, not from reasoning.** A `<tool_call>` emitted inside `<think>…</think>` is ignored by the tool parser (vLLM's reasoning/tool parsers cooperate so only post-`</think>` content is scanned). Don't expect calls made "while thinking" to fire.
- **Guided decoding is suppressed for GLM.** For `tool_choice: "required"` or a named tool, vLLM deliberately does **not** apply JSON structured-outputs/guided decoding, because that would force JSON output and conflict with GLM's XML syntax; the parser extracts from free-form XML instead.
- **`skip_special_tokens` must be off.** Although the tool/think tags are `special: false`, vLLM forces `skip_special_tokens = False` when tools are enabled (defensive against transformers 5.x detokenization changes) so the literal `<tool_call>`/`</tool_call>` text survives for the regex.
- **Streaming.** Long string arguments used to be buffered until the closing tag (vLLM issue #32829); the current parser re-parses the accumulated text each delta and emits only the diff, streaming incremental string content with an open-quote-then-fill strategy and holding back any partial trailing tag (`partial_tag_overlap`). The streamed tool name is the text before the first `\n` or `<arg_key>`. SGLang implements the same as an explicit XML→JSON state machine (`INIT → IN_KEY → WAITING_VALUE → IN_VALUE`). Malformed tails (a missing `</arg_value>` before `</tool_call>`) are closed off heuristically.
- **Lineage — GLM-4.5 vs GLM-4.6:** identical wire format and identical `chat_template.jinja` (same content hash); the same `glm45` parser serves both.
- **Lineage — GLM-4.7 / GLM-5 changed the format.** Newer models drop the structural newlines: the function name may sit **directly** before the first `<arg_key>` (no newline), zero-argument calls may be `<tool_call>func</tool_call>`, and parallel calls may be emitted **back-to-back with no separator** (`…</tool_call><tool_call>…`). These require the distinct `Glm47MoeModelToolParser` (vLLM, `structural_tag_model="glm_4_7"`) / `Glm47MoeDetector` (SGLang), whose `func_detail_regex` makes the newline and the argument section optional (`<tool_call>\s*(\S+?)\s*(<arg_key>.*)?</tool_call>`). Do **not** use a GLM-4.7 stream to validate a GLM-4.5 parser or vice versa.
## Sources
- Chat template (authoritative; rendered locally for the byte-exact streams), GLM-4.5 commit `cbb2c7c…`: https://huggingface.co/zai-org/GLM-4.5/resolve/main/chat_template.jinja — the `blob`/web path redirects to the model-card API; verified via the raw `resolve/main` cache.
- Identical GLM-4.6 template (same content hash, confirming shared format): https://huggingface.co/zai-org/GLM-4.6/resolve/main/chat_template.jinja
- Special-token IDs and `special` flags (`added_tokens_decoder`, `additional_special_tokens`): https://huggingface.co/zai-org/GLM-4.5/resolve/main/tokenizer_config.json
- Stop tokens (`eos_token_id = [151329, 151336, 151338]`): https://huggingface.co/zai-org/GLM-4.5/resolve/main/generation_config.json
- Model card (server flags `--tool-call-parser glm45 --reasoning-parser glm45`, `enable_thinking` switch, parser links): https://huggingface.co/zai-org/GLM-4.5
- vLLM GLM-4.5/4.6 tool parser (`Glm4MoeModelToolParser`: regexes, schema-driven string typing, JSON-string `arguments`, streaming, `skip_special_tokens`): https://github.com/vllm-project/vllm/blob/main/vllm/tool_parsers/glm4_moe_tool_parser.py
- vLLM GLM-4.7 tool parser (`Glm47MoeModelToolParser`: same-line name, optional/zero args): https://github.com/vllm-project/vllm/blob/main/vllm/tool_parsers/glm47_moe_tool_parser.py
- SGLang GLM-4.5/4.6 detector (`Glm4MoeDetector`: format docstring, XML→JSON state machine, argument typing): https://github.com/sgl-project/sglang/blob/main/python/sglang/srt/function_call/glm4_moe_detector.py
- SGLang GLM-4.7 detector (`Glm47MoeDetector`: newline-less / back-to-back calls): https://github.com/sgl-project/sglang/blob/main/python/sglang/srt/function_call/glm47_moe_detector.py
- vLLM tool-calling docs: https://docs.vllm.ai/en/latest/features/tool_calling/
+224
View File
@@ -0,0 +1,224 @@
# OpenAI Harmony response format
Harmony is the response format OpenAI trained its open-weight `gpt-oss` models on (`gpt-oss-20b`, `gpt-oss-120b`, released August 2025). It defines the conversation envelope, the multi-channel reasoning/answer separation, and the function-calling wire syntax. The models will not work correctly if prompted without it. The format deliberately mirrors the OpenAI *Responses* API (roles, channels, recipients) rather than the older Chat Completions shape.
Tokens are produced with the `o200k_harmony` encoding (the `o200k_base` BPE vocab plus a block of Harmony special tokens; see the table below). The reference renderer/parser is the Rust crate `openai-harmony` (Python bindings: `pip install openai-harmony`; encoding name `HarmonyEncodingName.HARMONY_GPT_OSS`).
You only deal with raw Harmony if you build your own inference loop. Served through an OpenAI-compatible endpoint the server handles it for you:
- **Ollama / LM Studio / HuggingFace**: Harmony is applied internally; you send normal OpenAI-style JSON.
- **vLLM**: `vllm serve openai/gpt-oss-120b --enable-auto-tool-choice --tool-call-parser openai --reasoning-parser openai_gptoss`. Note the tool-call parser flag is `openai` (not `harmony`). vLLM also exposes a Harmony-native path through the `/v1/responses` endpoint.
- **SGLang**: `python3 -m sglang.launch_server --model-path openai/gpt-oss-20b --reasoning-parser gpt-oss --tool-call-parser gpt-oss` (in NVIDIA Dynamo disaggregated mode: `--dyn-tool-call-parser harmony --dyn-reasoning-parser gpt_oss`).
The chat template shipped with the gpt-oss weights renders these same token sequences from the standard `messages`/`tools` arrays.
## Special tokens
All Harmony control tokens have the literal form `<|type|>` (ASCII pipes `|`, U+007C — no unicode variants). They are real single tokens in `o200k_harmony`, not text that is BPE-split. The structurally meaningful ones:
| Token (verbatim) | Token ID | Purpose |
| :--------------- | :------- | :------ |
| `<\|start\|>` | `200006` | Begins a message; immediately followed by the header (role, optional recipient/channel/content-type). |
| `<\|end\|>` | `200007` | Ends a fully-formed message. |
| `<\|message\|>` | `200008` | Header → content transition. Everything after it (until a stop/end token) is the message body. |
| `<\|channel\|>` | `200005` | Introduces the channel field of the header (`analysis` / `commentary` / `final`). |
| `<\|constrain\|>` | `200003` | Marks the content-type / constrained-decoding format in a tool-call header (e.g. `<\|constrain\|>json`). |
| `<\|return\|>` | `200002` | Stop token: the model finished its final answer. Decode-time only (see normalization note). |
| `<\|call\|>` | `200012` | Stop token: the model is emitting a tool call and wants it executed. |
`<|return|>` and `<|call|>` are the two valid generation stop tokens — halt inference on either.
The encoding also defines (same `o200k_harmony` block, IDs `199998`–`200013`) `<|startoftext|>` (199998), `<|endoftext|>` (199999), and reserved slots `<|reserved_200000|>`, `<|reserved_200001|>`, `<|reserved_200004|>`, `<|reserved_200009|>`–`<|reserved_200011|>`, `<|reserved_200013|>`, plus a bulk reserved range `<|reserved_200014|>`…`<|reserved_201088|>`. The renderer additionally knows the names `<|refusal|>`, `<|untrusted|>`, `<|end_untrusted|>`, `<|meta_end|>` but they are not part of the committed gpt-oss vocabulary and do not appear in normal traffic.
## Roles / channels / turn structure
**Message envelope.** Every message is:
```text
<|start|>{header}<|message|>{content}<|end|>
```
`{header}` always begins with the role and may carry an optional recipient (`to=...`), channel, and content-type. A completed message ends with `<|end|>`; an assistant message being generated ends instead with a stop token (`<|return|>` or `<|call|>`).
**Roles** (five). The instruction hierarchy used to resolve conflicts is `system` > `developer` > `user` > `assistant` > `tool`.
| Role | Purpose |
| :--- | :------ |
| `system` | Identity, knowledge cutoff / current date, reasoning effort, valid-channels declaration, built-in tools. NOT the user-facing "system prompt". |
| `developer` | The conventional "system prompt": instructions + the `# Tools` function declarations + (optional) structured-output schema. |
| `user` | End-user input. |
| `assistant` | Model output. Carries a channel and, for tool calls, a recipient. |
| `tool` | Output of an executed tool. The message's *author/role is the tool's own name* (e.g. `functions.get_current_weather`), not the literal word `tool`. |
**Channels** (assistant output only; the channel is mandatory on every assistant message):
| Channel | Purpose |
| :------ | :------ |
| `analysis` | Raw chain-of-thought (reasoning). Not held to the same safety bar as `final`; do not show to end users. Built-in `python`/`browser` calls usually go here. |
| `commentary` | Function tool calls, and user-visible "preambles" (action plans) before calling multiple tools. |
| `final` | The user-facing answer. |
**Reasoning effort** is set in the system message as `Reasoning: high` (or `medium` / `low`; default is medium). The model emits CoT into `analysis` and the answer into `final`.
**CoT carry-over rule.** On the next turn, drop prior `analysis` messages *if* the last assistant turn ended in a `final` message. The exception is an in-progress tool-calling turn: the `analysis` that preceded a tool call MUST be fed back in alongside the tool result so the model can continue its reasoning (the `openai-harmony` renderer does this via `RenderConversationConfig { auto_drop_analysis: true }`).
## Tool definitions
Function tools are advertised in the **developer** message under a `# Tools` section, inside a TypeScript-style `namespace functions { ... }`. (Built-in `browser`/`python` tools are instead declared in the **system** message under their own `# Tools` / `## browser` / `## python` headings.) The renderer converts each JSON Schema into a TS type with these rules:
- No-arg function → `type name = () => any;`
- With args → the single parameter is named `_` and its object type is inlined: `type name = (_: { ... }) => any;`
- Return type is always `any`.
- A property `description` becomes a `//` comment on the line *above* the field; a JSON Schema `title` renders as `// TITLE` followed by a `//` blank-comment line; `examples` render as `// Examples:` then `// - "value"` lines.
- Optional (non-`required`) fields get a trailing `?`. A `default` renders as a trailing `// default: <value>` comment; an `enum` becomes a `"a" | "b"` union; `oneOf` becomes a multi-line `|` union; JSON `integer` maps to TS `number`.
- One blank line separates function definitions; the block closes with `} // namespace functions`.
If the developer message has no instruction text, the `# Instructions` heading is omitted and the message is just the `# Tools` block. When any function is defined, the system message gains the routing line `Calls to these tools must go to the commentary channel: 'functions'.`
Verbatim developer-message example (instructions + two functions), exactly as the renderer emits it:
```text
<|start|>developer<|message|># Instructions
Use a friendly tone.
# Tools
## functions
namespace functions {
// Gets the location of the user.
type get_location = () => any;
// Gets the current weather in the provided location.
type get_current_weather = (_: {
// The city and state, e.g. San Francisco, CA
location: string,
format?: "celsius" | "fahrenheit", // default: celsius
}) => any;
// Gets the current weather in the provided list of locations.
type get_multiple_weathers = (_: {
// List of city and state, e.g. ["San Francisco, CA", "New York, NY"]
locations: string[],
format?: "celsius" | "fahrenheit", // default: celsius
}) => any;
} // namespace functions<|end|>
```
## Tool-call format
A function call is an **assistant** message on the **commentary** channel, addressed to the tool via recipient `to=functions.<name>`, with the JSON arguments as the body, terminated by the `<|call|>` stop token.
The recipient may appear in the *role section* or the *channel section* of the header — both are valid Harmony and the parser accepts either. The model commonly emits it in the channel section. The pi renderer omits the optional content-type marker:
```text
<|start|>assistant<|channel|>commentary to=functions.get_current_weather<|message|>{"location":"San Francisco, CA"}<|call|>
```
Some Harmony serializers include an explicit JSON content type and place the recipient in the role section instead:
```text
<|start|>assistant to=functions.get_current_weather<|channel|>commentary <|constrain|>json<|message|>{"location":"San Francisco, CA"}<|call|>
```
The arguments body is a raw JSON object. The optional `<|constrain|>json` content-type signals JSON (and is the hook for constrained/grammar-based decoding); the content-type may also be a bare word such as `code` (seen with built-in tools). Built-in tools differ only in channel and recipient: they typically render on `analysis`, with recipient `browser.search` / `browser.open` / `browser.find` or always `python`.
## Multiple / parallel tool calls
Harmony has no special "parallel" wrapper. Multiple calls are just multiple consecutive messages. The model may first emit an optional **preamble** — a *user-visible* assistant message on the `commentary` channel (unlike `analysis`, this is meant to be shown) — then one tool-call message per function. Each individual call still ends with its own `<|call|>` stop token, so a host that stops on `<|call|>` collects calls one at a time, executes, feeds the result back, and resumes:
```text
<|channel|>analysis<|message|>{reasoning}<|end|><|start|>assistant<|channel|>commentary<|message|>**Action plan**:
1. Generate an HTML file
2. Generate a JavaScript for the Node.js server
3. Start the server
---
Will start executing the plan step by step<|end|><|start|>assistant<|channel|>commentary to=functions.generate_file<|message|>{"template": "basic_html", "path": "index.html"}<|call|>
```
## Tool-result format
The executed tool's output is fed back as a message whose **author/role is the tool's name**, addressed back to the assistant (`to=assistant`), on the **commentary** channel, ending with `<|end|>`. This is the canonical (recommended) form:
```text
<|start|>functions.get_current_weather to=assistant<|channel|>commentary<|message|>{"sunny": true, "temperature": 20}<|end|>
```
The header ordering is `{toolname} to=assistant<|channel|>commentary`. Built-in tool results follow the same shape (e.g. `<|start|>browser.search to=assistant<|channel|>commentary<|message|>{"result": "https://openai.com/"}<|end|>`). The minimal form the renderer accepts when channel/recipient are not set on the message is just `<|start|>{toolname}<|message|>{output}<|end|>`, but emitting the full `to=assistant<|channel|>commentary` header is what the reference parser round-trips and is recommended. After appending the result, restart generation by emitting the next `<|start|>assistant`.
## End-to-end example
Complete multi-turn weather exchange: system + developer prompt → user question → assistant analysis CoT → assistant commentary tool call → tool result → assistant final answer. This is a single contiguous token stream (newlines inside headers are only between top-level messages for readability; in practice messages are concatenated with no separator).
```text
<|start|>system<|message|>You are ChatGPT, a large language model trained by OpenAI.
Knowledge cutoff: 2024-06
Current date: 2025-06-28
Reasoning: high
# Valid channels: analysis, commentary, final. Channel must be included for every message.
Calls to these tools must go to the commentary channel: 'functions'.<|end|><|start|>developer<|message|># Instructions
Use a friendly tone.
# Tools
## functions
namespace functions {
// Gets the current weather in the provided location.
type get_current_weather = (_: {
// The city and state, e.g. San Francisco, CA
location: string,
format?: "celsius" | "fahrenheit", // default: celsius
}) => any;
} // namespace functions<|end|><|start|>user<|message|>What is the weather like in SF?<|end|><|start|>assistant<|channel|>analysis<|message|>User wants the weather in San Francisco. Use get_current_weather.<|end|><|start|>assistant<|channel|>commentary to=functions.get_current_weather<|message|>{"location":"San Francisco, CA"}<|call|><|start|>functions.get_current_weather to=assistant<|channel|>commentary<|message|>{"sunny": true, "temperature": 20}<|end|><|start|>assistant<|channel|>final<|message|>It's sunny and about 20°C in San Francisco right now.<|return|>
```
Turn boundaries:
- The host stops generation at `<|call|>`, parses the `commentary` call, runs `get_current_weather`, and appends the `functions.get_current_weather to=assistant` result message.
- It then appends `<|start|>assistant` and resumes. The preceding `analysis` message is kept (the turn ended in a tool call, not a `final`), so the model can continue its reasoning.
- Generation stops at `<|return|>`. When this turn is persisted into history for a *later* turn, normalize the trailing `<|return|>` to `<|end|>` (see next note).
**`<|return|>` normalization.** `<|return|>` is a decode-time stop token only. When you store the assistant's reply into history for the next turn, replace the trailing `<|return|>` with `<|end|>` so every stored message is a well-formed `<|start|>{header}<|message|>{content}<|end|>`. (For supervised training targets, ending the example with `<|return|>` is correct.)
## OpenAI-compatible API mapping
When a server (vLLM/SGLang/Ollama) bridges Harmony to Chat Completions JSON:
- **`finish_reason`**: `tool_calls` when generation stopped on `<|call|>`; `stop` when it stopped on `<|return|>`.
- **`message.tool_calls[]`**: one entry per `commentary` `to=functions.*` call. `function.name` is the recipient with the `functions.` namespace stripped (`get_current_weather`). `function.arguments` is a **JSON string** (the verbatim `<|message|>` body), matching OpenAI semantics — not a parsed object.
- **`tool_call_id`**: Harmony has no native call ID. The server synthesizes one (e.g. `call_abc123`) and is responsible for correlating the follow-up `role:"tool"` message back to the Harmony tool-result envelope (recipient `to=functions.<name>` / call order).
- **Tool result messages** (`{"role":"tool","tool_call_id":...,"content":...}`) are rendered into `<|start|>{toolname} to=assistant<|channel|>commentary<|message|>{content}<|end|>`. The server maps `tool_call_id` → the original function name to build the `{toolname}` author.
- **Reasoning**: `analysis`-channel text is surfaced as `reasoning_content` (vLLM/SGLang) or as a `reasoning`/`thinking` field, and is generally not echoed back on subsequent requests. `final`-channel text is the normal `message.content`. `commentary` preambles, if surfaced, also map to assistant content.
- **`tools` / `tool_choice`** request fields are compiled by the chat template into the developer-message `namespace functions { ... }` block; the system message gains the commentary-routing line.
## Parsing notes & gotchas
- **Two stop tokens.** Always stop on both `<|return|>` and `<|call|>`. Stopping only on `<|return|>` will run past tool calls; stopping only on `<|end|>` is wrong for assistant generation.
- **Recipient position varies.** `to=functions.<name>` may be in the role section (`<|start|>assistant to=...<|channel|>commentary`) or the channel section (`<|channel|>commentary to=...`). A parser must accept both.
- **Channel is mandatory** on assistant messages; the system message even reminds the model ("Channel must be included for every message."). Missing-channel output is malformed.
- **Tool author, not `tool`.** The tool-result message's role is the tool's *name* (`functions.get_current_weather`), not the literal string `tool`. Splitting `functions.x` into namespace + function is the parser's job.
- **CoT dropping is conditional.** Drop `analysis` only when the previous assistant turn ended on `final`. Dropping the `analysis` that immediately precedes a `<|call|>` breaks multi-step tool reasoning.
- **`arguments` is a string.** Do not double-encode. The body after `<|message|>` is already serialized JSON; pass it through as the `arguments` string.
- **Content-type variants.** `<|constrain|>json` is optional. If present, it is metadata, not a guarantee of valid JSON. Enforce JSON validity with constrained decoding / your own grammar — the prompt format alone does not guarantee schema adherence (same caveat applies to structured-output `# Response Formats`).
- **Streaming.** Use a stateful parser (the library ships `StreamableParser`) so partial UTF-8 and the header/channel/recipient/content-type fields are reconstructed incrementally; a naive substring scan mishandles multi-byte splits and the optional header fields. `parse_messages_from_completion_tokens` takes `strict=True|False` — `strict=False` tolerates some malformed headers. Do not pass the trailing stop token into the parser.
- **Encoding.** Use `o200k_harmony` (the `o200k_base` ranks plus the Harmony specials above). Treat the `<|...|>` tokens as atomic special tokens during both encode and decode; encoding them as ordinary text yields different ranks and corrupts the stream.
## Sources
- OpenAI Cookbook — OpenAI harmony response format: https://cookbook.openai.com/articles/openai-harmony
- openai/harmony renderer (README): https://github.com/openai/harmony
- openai/harmony canonical format guide: https://raw.githubusercontent.com/openai/harmony/main/docs/format.md
- openai/harmony special-token registry (`o200k_harmony` IDs): https://raw.githubusercontent.com/openai/harmony/main/src/tiktoken_ext/public_encodings.rs
- openai/harmony renderer/parser tests and schema→TS logic: https://raw.githubusercontent.com/openai/harmony/main/src/tests.rs , https://raw.githubusercontent.com/openai/harmony/main/src/encoding.rs
- openai/harmony test fixtures (verbatim rendered streams): `test-data/test_render_functions_with_parameters.txt`, `test-data/test_does_not_drop_if_ongoing_analysis.txt`, `test-data/test_tool_response_parsing.txt`, `test-data/test_streamable_parser.txt`, `test-data/test_browser_and_function_tool.txt` (https://github.com/openai/harmony/tree/main/test-data)
- vLLM tool calling / gpt-oss parser flags: https://docs.vllm.ai/en/latest/features/tool_calling/
- SGLang gpt-oss usage (`--tool-call-parser gpt-oss`): https://docs.sglang.io/basic_usage/gpt_oss.html
+182
View File
@@ -0,0 +1,182 @@
# Kimi K2 tool-calling format
Native tool-calling convention of Moonshot AI's **Kimi K2** family (`moonshotai/Kimi-K2-Instruct` and `-Base`, `model_type: "kimi_k2"`, 1T-param MoE). It is a ChatML-like envelope built on a TikToken tokenizer (160K vocab): every turn is `<|im_{class}|>{name}<|im_middle|>{body}<|im_end|>`, and tool calls are emitted inside the assistant turn wrapped by a dedicated `<|tool_calls_section_begin|>…<|tool_calls_section_end|>` block. All control tokens are plain ASCII `<|…|>` forms (no fullwidth/unicode variants, unlike DeepSeek). An inference server turns the raw stream into OpenAI-style `tool_calls` with a parser: vLLM and SGLang both expose `--tool-call-parser kimi_k2` (vLLM additionally requires `--enable-auto-tool-choice`). The chat template (a standalone `chat_template.jinja` since the 2025.8.11 update) injects the tool schemas and renders the per-turn markers.
This document was verified against the model card, the official `docs/tool_call_guidance.md` and `docs/deploy_guidance.md` (GitHub `MoonshotAI/Kimi-K2`), the raw `chat_template.jinja` and `tokenizer_config.json` from the HF repo (rendered locally for the byte-exact streams below), and the vLLM `kimi_k2` tool parser source.
## Special tokens
The five tool-call markers required for manual parsing, plus the ChatML envelope markers. Token IDs are from `tokenizer_config.json` (`added_tokens_decoder`).
| Token (verbatim) | ID | Purpose |
|---|---|---|
| `<\|tool_calls_section_begin\|>` | 163595 | Opens the tool-call section inside an assistant turn |
| `<\|tool_call_begin\|>` | 163597 | Opens one individual tool call |
| `<\|tool_call_argument_begin\|>` | 163598 | Separates the tool-call ID from its JSON arguments |
| `<\|tool_call_end\|>` | 163599 | Closes one individual tool call |
| `<\|tool_calls_section_end\|>` | 163596 | Closes the tool-call section |
| `<\|im_system\|>` | 163594 | Start marker for system-class turns (`system`, `tool`, `tool_declare`) |
| `<\|im_user\|>` | 163587 | Start marker for a user turn |
| `<\|im_assistant\|>` | 163588 | Start marker for an assistant turn |
| `<\|im_middle\|>` | 163601 | Separates the role/name header from the message body |
| `<\|im_end\|>` | 163586 | Ends any turn |
| `[BOS]` | 163584 | Sequence-begin token (see notes; not emitted by the chat template) |
| `[EOS]` | 163585 | Sequence-end token |
Notes on exactness:
- The five tool tokens use ASCII pipe `|` (U+007C) and underscores; reproduce them exactly. There are no fullwidth pipe (`|`) or `▁` variants in Kimi K2.
- `<|im_middle|>` is the only envelope token whose ID (163601) is out of sequence with the others (163586–163599); a `163600` slot is unused.
- Image inputs render via a content macro as the literal sequence `<|media_start|>image<|media_content|><|media_pad|><|media_end|>`. These media markers appear in the template but are **not** registered in `added_tokens_decoder`, so they tokenize as ordinary text rather than single special tokens. They are irrelevant to text tool calling and are listed here only for completeness.
## Roles / channels / turn structure
Kimi K2 uses a ChatML-style envelope. Every message is rendered as:
```text
<|im_{class}|>{name}<|im_middle|>{body}<|im_end|>
```
- There are exactly **three** start-marker tokens, chosen by `role`:
- `user` → `<|im_user|>`
- `assistant` → `<|im_assistant|>`
- everything else (`system`, `tool`, and the synthetic `tool_declare`) → `<|im_system|>`
- The `{name}` segment between the marker and `<|im_middle|>` is `message.name or message.role`. This is the only "channel"/sub-role label Kimi K2 has. For ordinary turns it is literally `system`, `user`, or `assistant`; for a tool-result turn it is the tool's `name` (the function name) when supplied, otherwise `tool`; for the tool-schema turn it is the literal `tool_declare`.
- `<|im_end|>` terminates every turn. The chat template does **not** emit `[BOS]`/`[EOS]`; turn boundaries are purely `<|im_*|>` markers (the tokenizer is TikToken-based with `add_bos_token`/`add_eos_token` unset, and the manual-parse flow feeds the rendered template straight to `/completions`).
- **Default system prompt:** if the first message is not a `system` message, the template injects `<|im_system|>system<|im_middle|>You are Kimi, an AI assistant created by Moonshot AI.<|im_end|>` before the first turn.
- **Generation prompt:** with `add_generation_prompt=True` the template ends with `<|im_assistant|>assistant<|im_middle|>`, and the model generates from there.
- **Thinking/reasoning:** `Kimi-K2-Instruct` is a "reflex-grade" model with no long thinking, so there is no reasoning channel in this format. (Thinking variants are handled separately — vLLM ships a distinct `kimi_k2` reasoning parser keyed on a `</think>` token — but that is out of scope for the Instruct tool-call format documented here.)
## Tool definitions
Available tools are advertised in a single dedicated turn placed at the very top of the prompt (before any system/user turn), using the synthetic `tool_declare` sub-role under the `<|im_system|>` marker:
```text
<|im_system|>tool_declare<|im_middle|>{TOOLS_JSON}<|im_end|>
```
`{TOOLS_JSON}` is the standard OpenAI-style `tools` array serialized to JSON with **compact separators** `(',', ':')` (no spaces). The array elements are passed through verbatim, i.e. each is `{"type":"function","function":{"name":…,"description":…,"parameters":{…}}}` with a JSON-Schema `parameters` object. Example (single tool, exactly as emitted):
```text
<|im_system|>tool_declare<|im_middle|>[{"type":"function","function":{"name":"get_weather","description":"Get weather information. Call this tool when the user needs to get weather information","parameters":{"type":"object","required":["city"],"properties":{"city":{"type":"string","description":"City name"}}}}}]<|im_end|>
```
The `tool_declare` turn is rendered only when `tools` is non-empty.
## Tool-call format
When the model decides to call a function, it emits — inside the assistant turn, after any natural-language content — a tool-calls section. Minimal single call (this is the assistant generation that follows `<|im_assistant|>assistant<|im_middle|>`):
```text
<|tool_calls_section_begin|><|tool_call_begin|>functions.get_weather:0<|tool_call_argument_begin|>{"city": "Beijing"}<|tool_call_end|><|tool_calls_section_end|>
```
Anatomy of one call:
```text
<|tool_call_begin|> functions.{func_name}:{idx} <|tool_call_argument_begin|> {JSON arguments} <|tool_call_end|>
```
- The token between `<|tool_call_begin|>` and `<|tool_call_argument_begin|>` is the **tool-call ID**, with the fixed form `functions.{func_name}:{idx}`.
- `functions.` is a literal prefix (it is not derived from the tool schema).
- `{func_name}` is the called function's name; the function name is recovered by parsing it back out of this ID, not from a separate field.
- `{idx}` is the **0-based call index** within the current assistant turn (`0` for the first call, `1` for the second, …).
- After `<|tool_call_argument_begin|>` comes the raw JSON arguments object (e.g. `{"city": "Beijing"}`), terminated by `<|tool_call_end|>`.
- All calls of the turn live between one `<|tool_calls_section_begin|>` / `<|tool_calls_section_end|>` pair. Any assistant text content precedes `<|tool_calls_section_begin|>`.
- The whole assistant turn is still closed by `<|im_end|>` and the completion's `finish_reason` becomes `tool_calls`.
## Multiple / parallel tool calls
Two or more calls in one turn are emitted as consecutive `<|tool_call_begin|>…<|tool_call_end|>` blocks inside a single section, with the index incrementing per call. Raw assistant emission for two parallel calls:
```text
<|tool_calls_section_begin|><|tool_call_begin|>functions.get_weather:0<|tool_call_argument_begin|>{"city": "Beijing"}<|tool_call_end|><|tool_call_begin|>functions.get_weather:1<|tool_call_argument_begin|>{"city": "Shanghai"}<|tool_call_end|><|tool_calls_section_end|>
```
Note the IDs `functions.get_weather:0` and `functions.get_weather:1` — same function, distinct trailing index. The index is per-turn (it resets to `0` in the next assistant turn).
## Tool-result format
Tool execution results are fed back as a turn with `role: "tool"`. Because `tool` is not `user`/`assistant`, it renders under the `<|im_system|>` marker; the sub-role label is the message's `name` (the function name) when present, else `tool`. The body is a literal `## Return of {tool_call_id}` header line followed by the result content:
```text
<|im_system|>get_weather<|im_middle|>## Return of functions.get_weather:0
{"weather": "Sunny"}<|im_end|>
```
- `{tool_call_id}` echoes the exact ID from the originating call (`functions.get_weather:0`), which is how the model correlates a result with the call that produced it.
- The result `content` is inserted verbatim on the line after the header; callers typically pass a JSON string (e.g. `json.dumps(tool_result)`).
- If the `tool` message omits `name`, the envelope becomes `<|im_system|>tool<|im_middle|>## Return of …`.
## End-to-end example
A complete multi-turn weather exchange. These are the exact rendered streams (system + user supplied explicitly; line breaks inside a turn are literal, turns are otherwise contiguous).
**Stage 1 — prompt fed to the model** (`tools` set, `add_generation_prompt=True`):
```text
<|im_system|>tool_declare<|im_middle|>[{"type":"function","function":{"name":"get_weather","description":"Get weather information. Call this tool when the user needs to get weather information","parameters":{"type":"object","required":["city"],"properties":{"city":{"type":"string","description":"City name"}}}}}]<|im_end|><|im_system|>system<|im_middle|>You are Kimi, an AI assistant created by Moonshot AI.<|im_end|><|im_user|>user<|im_middle|>What's the weather like in Beijing today? Use the tool to check.<|im_end|><|im_assistant|>assistant<|im_middle|>
```
**Assistant generation** (model output; server reports `finish_reason: "tool_calls"`):
```text
<|tool_calls_section_begin|><|tool_call_begin|>functions.get_weather:0<|tool_call_argument_begin|>{"city": "Beijing"}<|tool_call_end|><|tool_calls_section_end|><|im_end|>
```
**Stage 2 — prompt for the next turn**, after appending the assistant tool-call turn and the tool result turn (`add_generation_prompt=True`):
```text
<|im_system|>tool_declare<|im_middle|>[{"type":"function","function":{"name":"get_weather","description":"Get weather information. Call this tool when the user needs to get weather information","parameters":{"type":"object","required":["city"],"properties":{"city":{"type":"string","description":"City name"}}}}}]<|im_end|><|im_system|>system<|im_middle|>You are Kimi, an AI assistant created by Moonshot AI.<|im_end|><|im_user|>user<|im_middle|>What's the weather like in Beijing today? Use the tool to check.<|im_end|><|im_assistant|>assistant<|im_middle|><|tool_calls_section_begin|><|tool_call_begin|>functions.get_weather:0<|tool_call_argument_begin|>{"city": "Beijing"}<|tool_call_end|><|tool_calls_section_end|><|im_end|><|im_system|>get_weather<|im_middle|>## Return of functions.get_weather:0
{"weather": "Sunny"}<|im_end|><|im_assistant|>assistant<|im_middle|>
```
**Final assistant generation** (model produces natural-language answer terminated by `<|im_end|>`; `finish_reason: "stop"`):
```text
It's sunny in Beijing today.<|im_end|>
```
## OpenAI-compatible API mapping
With a server parser active (`--tool-call-parser kimi_k2`), the raw stream maps onto the Chat Completions shape as follows:
- `choices[].finish_reason` = `"tool_calls"` when the turn contained a tool-calls section (otherwise `"stop"`).
- `choices[].message.tool_calls[]` — one entry per `<|tool_call_begin|>…<|tool_call_end|>` block:
- `.id` = the raw call ID verbatim, e.g. `"functions.get_weather:0"`.
- `.type` = `"function"`.
- `.function.name` = the function name parsed out of the ID. vLLM computes `id.split(":")[0].split(".")[-1]` → `"get_weather"`.
- `.function.arguments` = a **JSON string** (the raw text captured between `<|tool_call_argument_begin|>` and `<|tool_call_end|>`), e.g. `"{\"city\": \"Beijing\"}"`. Clients `json.loads()` it before use.
- Tool results are sent back as messages of the form:
```json
{"role": "tool", "tool_call_id": "functions.get_weather:0", "name": "get_weather", "content": "{\"weather\": \"Sunny\"}"}
```
`tool_call_id` must equal the `id` returned for the call; `name` becomes the `<|im_system|>{name}<|im_middle|>` sub-role; `content` becomes the body after `## Return of …`.
- Streaming: deltas arrive as `choices[].delta.tool_calls[]` with an `index`; the function `name`/`id` stream once the call header is complete, then `function.arguments` streams as incremental string fragments to be concatenated (standard OpenAI tool-call streaming assembly).
Moonshot's hosted API (`platform.moonshot.ai`) exposes both OpenAI- and Anthropic-compatible endpoints; the Anthropic-compatible one scales temperature as `real_temperature = request_temperature * 0.6`. Recommended sampling temperature for `Kimi-K2-Instruct` is `0.6`.
## Parsing notes & gotchas
- **ID → name parsing differs between references.** The official `tool_call_guidance.md` extracts the name with `function_id.split('.')[1].split(':')[0]`, which assumes the ID is exactly `functions.{name}` with no extra dots. vLLM uses the more robust `function_id.split(":")[0].split(".")[-1]` (takes the last dot-segment before `:{idx}`). Prefer the vLLM form so function names containing `.` are handled.
- **Extraction regexes differ too.** Guidance: `<\|tool_call_begin\|>\s*(?P<tool_call_id>[\w\.]+:\d+)\s*<\|tool_call_argument_begin\|>\s*(?P<function_arguments>.*?)\s*<\|tool_call_end\|>`. vLLM: ID class is `[^<]+:\d+` and the argument body uses a negative lookahead `(?:(?!<\|tool_call_begin\|>).)*?` so adjacent calls aren't merged. Both run with `DOTALL`.
- **`skip_special_tokens` must be False.** The parser depends on the literal marker text surviving detokenization; vLLM forces `skip_special_tokens = False` when tools are enabled and `tool_choice != "none"`. If markers are stripped, no tool call is detected.
- **Arguments are unvalidated raw text.** Whatever the model emits between the argument marker and `<|tool_call_end|>` is passed straight through as the `arguments` string; it must be valid JSON for downstream `json.loads`, and the model can emit malformed/truncated JSON. Validate before executing.
- **Index semantics.** `{idx}` is the per-turn call counter starting at `0`; it is not a global counter and resets each assistant turn. Do not assume IDs are unique across turns — disambiguate by turn when persisting history.
- **Streaming marker splits.** Section and call markers can be split across token boundaries. vLLM holds back any trailing suffix that partially matches a marker (`partial_tag_overlap`) to avoid leaking marker bytes into streamed content, and only streams a call's name once its header is fully received.
- **`finish_reason` varies by engine.** The official guide explicitly warns the terminal `finish_reason` for tool calls "may vary across different engines"; loop on `finish_reason == "tool_calls"` but be defensive.
- **Engine fallback.** Kimi K2 reuses the DeepSeek-V3 architecture; `config.json` sets `model_type: "kimi_k2"` so engines apply the right parser. If you force `model_type: "deepseek_v3"` as a compatibility workaround, no native Kimi tool parser is available and you must parse the `<|tool_calls_section_*|>` markers manually.
- **Parser availability.** vLLM ships both a Python (`KimiK2ToolParser`) and a newer Rust tool parser; SGLang implements its own `kimi_k2` parser. All key off the same five markers and the `functions.{name}:{idx}` ID convention documented here.
- **Whitespace artifact.** When no `system` message is supplied, the template injects the default system prompt and a small `\n ` (newline + two spaces) can appear before the first `<|im_user|>` marker. It is harmless (tokenizes around the markers), but supplying an explicit system message yields the clean streams shown above.
## Sources
- Model card (Tool Calling section, OpenAI-style example, deployment/API notes): https://huggingface.co/moonshotai/Kimi-K2-Instruct
- Official tool-call guidance (markers, ID convention, manual parser, `extract_tool_call_info`): https://raw.githubusercontent.com/MoonshotAI/Kimi-K2/main/docs/tool_call_guidance.md (the HF `resolve`/`blob` paths redirected to the model card; verified against this GitHub raw file)
- Deployment guide (`--tool-call-parser kimi_k2`, `--enable-auto-tool-choice`, SGLang flag, `model_type` fallback): https://raw.githubusercontent.com/MoonshotAI/Kimi-K2/main/docs/deploy_guidance.md
- Chat template (`chat_template.jinja`, rendered locally for byte-exact streams): https://huggingface.co/moonshotai/Kimi-K2-Instruct/resolve/main/chat_template.jinja
- Tokenizer config (special-token IDs in `added_tokens_decoder`): https://huggingface.co/moonshotai/Kimi-K2-Instruct/resolve/main/tokenizer_config.json
- vLLM `kimi_k2` tool parser (markers, regex, name-parsing, `skip_special_tokens`, streaming): https://github.com/vllm-project/vllm/blob/main/vllm/tool_parsers/kimi_k2_tool_parser.py
- vLLM PR adding the parser: https://github.com/vllm-project/vllm/pull/20789
- vLLM tool-calling docs: https://docs.vllm.ai/en/latest/features/tool_calling/
+276
View File
@@ -0,0 +1,276 @@
# pi-native tool-call format
The **pi-native** format is the tool-call serialization used by the omp / pi coding agent. Unlike the JSON-in-a-tag conventions (Hermes/Qwen, Harmony) and unlike the fully separate JSON content-block channel (Anthropic Messages API), pi-native serializes each call as an **XML-flavored block** whose tag carries the tool name — `<call:NAME>…</call:NAME>` — and whose arguments are child elements named after the parameters. It is **schema-driven**: the tool's JSON Schema decides how each value is typed (string vs number vs object vs array) and which compact spellings are legal.
This document is a **specification** of the format (it is the contract a renderer must emit and a parser must accept), not a reverse-engineering of trained model weights. It is designed around four goals:
- **Token economy** — the common cases (a single scalar argument; a single string payload) collapse to one short line.
- **Verbatim payloads** — a large multi-line string argument (a patch body, a file's contents, a shell script) is carried **raw**, with no JSON string-escaping and no entity-encoding, terminated by the call's own unique closing tag.
- **Human legibility** — a call reads like the function it denotes; nesting maps to nesting.
- **Lenient parsing** — the tags are plain text matched by a tolerant parser (regex / streaming state machine), not a strict XML parser; output is *not* required to be well-formed XML.
Scope: pi-native specifies only the **tool-call (and argument) serialization**. It is **envelope-agnostic** — the `<call:…>` blocks are emitted as ordinary assistant text and embed unchanged in any conversation envelope (ChatML, Harmony, the Anthropic two-role shape, …). Reasoning channels, role markers, and result delivery are the host envelope's concern; the one envelope-level requirement pi-native imposes is in [Tool-result correlation](#tool-result-correlation).
Lineage: the attribute spelling (`<call:read path="…"/>`) follows Anthropic's modern attribute XML (`<invoke name="…">` / `<parameter name="…">`); the schema-driven **unquoted** value rule (a bare string value carries no quotes; non-strings are JSON) follows GLM-4.5's `<arg_value>` convention. pi-native folds both into one recursive, name-on-the-tag grammar and adds the verbatim **inline body** for bulk string arguments. See [`anthropic.md`](./anthropic.md) and [`glm-4.5.md`](./glm-4.5.md) in this folder.
## Structural tags
pi-native has **no special tokens**. Every marker is plain UTF-8 text that BPE-splits like any other text and survives detokenization unchanged; a parser matches the tags as literal substrings (and MUST work even when the surrounding stream is not valid XML). All brackets are ASCII `<` `>` `/` and the literal colon `:`. There are no namespaces, no `<?xml?>` prolog, no entity expansion, and no CDATA sections.
| Tag (verbatim) | Role |
|---|---|
| `<call:NAME …>` … `</call:NAME>` | One tool call. `NAME` is the tool/recipient name; it is repeated on the closing tag. |
| `<call:NAME …/>` | Self-closing tool call (all arguments supplied as attributes). |
| `<KEY>` … `</KEY>` | One argument (or one nested field). `KEY` is the parameter name. |
| `<KEY …/>` | Self-closing argument: an object-valued field whose scalar sub-fields are attributes, or an empty value. |
| `KEY="…"` / `KEY='…'` / `KEY=…` | An attribute: a scalar field given inline on a tag. Quotes are **delimiters, not type markers** (see [value coercion](#value-coercion)). |
Names (tool names and parameter names) match `^[A-Za-z_][A-Za-z0-9_-]*$`. The `call:` prefix is a literal four-character marker plus the colon; the colon is what distinguishes a call block from a nested argument element of the same name.
## Tool-call forms
A single call has three interchangeable surface forms. Which forms are legal for a given tool is decided by its parameter schema; a renderer SHOULD pick the most compact legal form, and a parser MUST accept all three.
### 1. Element form (canonical, fully general)
Each top-level argument is a child element named after the parameter; the element body is the value. This form expresses every schema — scalars, strings, arrays, and nested objects:
```text
<call:read>
<path>src/server/auth.ts</path>
<offset>50</offset>
</call:read>
```
→ `read({ "path": "src/server/auth.ts", "offset": 50 })` (`offset` is JSON because the schema types it as a number; `path` is a verbatim string).
### 2. Attribute form (compact scalars)
When the arguments being passed are **top-level scalars** (string, number, integer, boolean, null), they MAY be written as attributes on the call tag. With every argument as an attribute the tag is self-closing:
```text
<call:read path="src/server/auth.ts"/>
```
→ `read({ "path": "src/server/auth.ts" })`.
Attributes and child elements MAY be combined on a non-self-closing call tag — attributes carry the scalars, child elements carry anything structured:
```text
<call:read path="src/server/auth.ts">
<offset>50</offset>
</call:read>
```
An attribute whose value cannot be represented as a scalar (an object or array argument) MUST use the element form instead — there is no attribute spelling for structured values on a call tag.
### 3. Inline-body form (verbatim string payload)
When the tool's parameters are **all strings** — most often a single string parameter — the call body MAY be the argument value written **verbatim**, with no child element tags:
```text
<call:edit>
*** Begin Patch
@@ src/server/auth.ts
- return user;
+ return user ?? null;
*** End Patch
</call:edit>
```
→ `edit({ "input": "*** Begin Patch\n@@ src/server/auth.ts\n- return user;\n+ return user ?? null;\n*** End Patch" })`.
Rules for the inline body:
- The body fills the **first parameter not already supplied by an attribute**. With no attributes that is simply the first parameter (the "first argument verbatim").
- It is permitted only when that target parameter is **string**-typed (the enabling condition "the type only contains string arguments"); any *other* parameters set on the same call MUST be scalars given as attributes.
- The value is captured **verbatim** up to the call's own closing tag `</call:NAME>`. No JSON escaping, no entity decoding. The body MAY freely contain `<`, `>`, `&`, quotes, JSON, even other `<call:…>`-looking text — the only sequence it MUST NOT contain is the literal closer `</call:NAME>`. Because that closer carries the tool name, collisions are far rarer than with a short generic delimiter.
- Whitespace: a single newline immediately after the opening `>` and a single newline immediately before the closing `</` are treated as block delimiters and are **not** part of the value; all other whitespace (indentation, blank lines, trailing spaces) is preserved exactly.
A multi-string tool can still use the inline body for its bulk argument by passing the others as attributes — the body then fills the first parameter left unset:
```text
<call:write path="notes/todo.md">
# TODO
- ship pi-native parser
</call:write>
```
→ `write({ "path": "notes/todo.md", "content": "# TODO\n- ship pi-native parser" })` (here `path` is given by attribute, so the body fills the next string parameter, `content`).
Inline-eligible tools may always fall back to the element form; `<call:edit><input>…</input></call:edit>` and the inline `<call:edit>…</call:edit>` are equivalent.
## Value model
The body of a call (and of any nested element) maps to JSON by a single recursive rule set. **Typing is driven by the parameter's JSON Schema**; the parser only falls back to syntactic heuristics when no schema is available.
### Value coercion
Let `coerce(text, type)` produce the JSON value for a captured scalar `text`:
- `type == "string"` → the value is `text`, **verbatim** (never JSON-parsed, never unquoted). This is why `<path>4</path>` for a string parameter is the string `"4"`, and a Windows path `C:\new\tab` survives intact.
- `type` is a non-string scalar (`number` / `integer` / `boolean` / `null`) → `JSON.parse(text)` (so `<offset>50</offset>` → `50`, `<recursive>true</recursive>` → `true`).
- `type` unknown (no schema) → **best-effort JSON coercion**: try `JSON.parse(text)`; on success use the parsed value (number, boolean, null, quoted-string, object, or array); on failure treat `text` as a literal string. So a bare `4` becomes the number `4`, `foo.ts` (not valid JSON) becomes `"foo.ts"`.
The same `coerce` applies to **attribute values** after the surrounding quotes (if any) are stripped — the quotes are XML delimiters only. Hence both spellings below are identical, and both yield the **number** `4` (not the string `"4"`) when `y` is untyped/numeric:
```text
<object y=4/> → { "object": { "y": 4 } }
<object y="4"/> → { "object": { "y": 4 } }
```
Consequence to internalize: under loose/no schema, quoting does **not** force a string — `"4"` still coerces to `4`. To carry a numeric-looking value *as a string*, the parameter MUST be `string`-typed in the schema (then the verbatim rule keeps `"4"` → `"4"`). Unquoted attribute values run until whitespace or the closing `>` / `/>`; spaces around `=` are tolerated; a bare attribute with no `=value` (e.g. `<call:tool dry_run/>`) denotes boolean `true`.
### Scalars and strings
A scalar argument is one element (or one attribute). String values are unquoted and verbatim; non-string scalars are JSON literals:
```text
<call:bash command="ls -la" timeout=30/>
```
→ `bash({ "command": "ls -la", "timeout": 30 })`.
### Arrays — repeat the element
An array-typed field is expressed by **repeating** its element; each occurrence contributes one item, in order:
```text
<list>x</list>
<list>y</list>
```
→ `"list": ["x", "y"]`.
A field the schema types as an **array always yields an array, even for a single occurrence** — so one `<list>x</list>` under an array-typed `list` is `["x"]`, not `"x"`. When no schema is available the parser falls back to a count heuristic: a name appearing **2+ times among its siblings** is an array; a name appearing **once** is a scalar (so schema typing is the only way to express a one-element array with the heuristic alone). Item values coerce by the array's item type (`<ports>80</ports><ports>443</ports>` → `[80, 443]` for a `number[]`); arrays of objects repeat a nested block (see below). There is no attribute spelling for an array (attributes cannot repeat) — arrays require element form.
### Objects — a nested block
An object-typed field opens its own block and follows the **same rules recursively**: its child elements become its properties, repeated children become arrays, and nested object children open further blocks.
```text
<object>
<list>x</list>
</object>
```
→ `"object": { "list": ["x"] }` (with `object` typed object and `list` typed array).
An object's **scalar** sub-fields MAY instead be written as attributes — `<object y=4/>` is shorthand for `<object><y>4</y></object>`. Attributes and child elements may be combined on the same object element (attributes for scalars, children for structured sub-fields). An empty object is `<object/>` or `<object></object>` → `{}`.
### Recursion
The call body, an object element's body, and an array item's body are all parsed by the identical procedure. Parsing element `E` (tag = field name `F`, schema type `T`):
1. Gather `E`'s attributes → scalar properties via `coerce`.
2. Determine `E`'s body shape from `T` (or, with no schema, from whether the body's first non-whitespace content is a child tag):
- `T` object → properties from child elements (+ the attributes from step 1).
- `T` array (item type `Ti`) → collect **all** siblings named `F`; each occurrence is one item parsed as `Ti`.
- `T` scalar/string → the body is captured text; value = `coerce(text, T)`.
3. The call itself is element `E` with no enclosing key: its attributes + child elements **are** the arguments object directly (the tool name on `<call:NAME>` is the recipient, not a key).
## Multiple / parallel tool calls
There is no wrapper element around a set of calls. Parallel calls are simply **consecutive `<call:…>` blocks** in one assistant turn (separated by whitespace/newlines; interleaved prose is allowed and is ordinary content):
```text
<call:read path="src/a.ts"/>
<call:read path="src/b.ts"/>
```
A parser returns these as `tool_calls[0]`, `tool_calls[1]`, … in emission order. The host executes them and returns one result per call, in the same order (see correlation, next).
## Tool definitions and schema dependence
pi-native does not prescribe how tools are advertised; a host typically lists them as JSON Schema, exactly as the OpenAI / Anthropic / Hermes families do. What pi-native **requires** is that the parser have access to each tool's parameter schema, because the schema is what disambiguates:
- string (verbatim, unquoted) vs other scalar (JSON) values;
- a one-element array vs a scalar (a single `<list>…</list>`);
- which body shape (text vs nested members) a non-self-closing element carries;
- whether the inline-body form is legal (first unset parameter is a string).
Without a schema the parser MUST degrade gracefully to the syntactic fallbacks named above (JSON-coerce scalars; repetition-counts for arrays; child-tag presence for object bodies). The fallbacks are lossy at exactly the ambiguous points the schema would resolve, so production hosts SHOULD always supply the schema.
## Tool-result correlation
pi-native calls carry **no per-call wire id** (like GLM and Qwen, unlike Anthropic's `toolu_…`). Results are therefore correlated to calls **positionally, by emission order**: the host delivers tool outputs in the same order the `<call:…>` blocks appeared, using whatever its envelope provides for tool output (a `tool`/`user` turn, a Harmony tool message, an Anthropic `tool_result` block, …). When a transport requires an id (e.g. an OpenAI-compatible bridge), the host synthesizes one and maintains the call↔result mapping itself; the id never appears in the pi-native text.
## End-to-end example
A short agent turn exercising all three forms plus nesting. Schemas in play: `read(path: string, offset?: number)`, `bash(command: string, timeout?: number)`, `edit(input: string)`, and a synthetic `configure(object: { list: string[]; y?: number })`.
```text
I'll inspect the file, run the tests, then apply the fix.
<call:read path="src/server/auth.ts"/>
<call:bash command="bun test src/server/auth.test.ts" timeout=120/>
<call:configure>
<object y=4>
<list>alpha</list>
<list>beta</list>
</object>
</call:configure>
<call:edit>
*** Begin Patch
@@ src/server/auth.ts
- return user;
+ return user ?? null;
*** End Patch
</call:edit>
```
Parses to four calls, in order:
```json
[
{ "name": "read", "arguments": { "path": "src/server/auth.ts" } },
{ "name": "bash", "arguments": { "command": "bun test src/server/auth.test.ts", "timeout": 120 } },
{ "name": "configure", "arguments": { "object": { "y": 4, "list": ["alpha", "beta"] } } },
{ "name": "edit", "arguments": { "input": "*** Begin Patch\n@@ src/server/auth.ts\n- return user;\n+ return user ?? null;\n*** End Patch" } }
]
```
Note: `timeout=120` and `y=4` are JSON numbers (numeric/untyped scalars), `path` and the `list` items are verbatim strings (string-typed), `object` opens a nested block whose `y` rides as an attribute while `list` repeats into an array, and the `edit` body is captured verbatim up to `</call:edit>` despite containing `@@`, `-`/`+`, and other non-XML text.
## Grammar (lenient EBNF)
This is the shape a tolerant parser accepts; it is intentionally looser than XML (mismatched-but-recoverable tails are closed heuristically — see gotchas).
```ebnf
stream ::= ( text | call )*
call ::= self-call | block-call
self-call ::= "<call:" Name attr* ws? "/>"
block-call ::= "<call:" Name attr* ">" call-body "</call:" Name ">"
call-body ::= members | inline-text ; inline-text only if first param is string
members ::= ( ws | element )*
element ::= self-element | block-element
self-element ::= "<" Name attr* ws? "/>" ; object via attrs, or empty value
block-element::= "<" Name attr* ">" ( members | scalar-text ) "</" Name ">"
attr ::= ws Name ( ws? "=" ws? attr-val )? ; bare Name → boolean true
attr-val ::= '"' dq-chars '"' | "'" sq-chars "'" | bareword
Name ::= [A-Za-z_] [A-Za-z0-9_-]*
scalar-text ::= < any chars up to the matching close tag, verbatim >
inline-text ::= < any chars up to "</call:" Name ">", verbatim >
```
## Parsing notes & gotchas
- **Schema decides string-vs-JSON.** A `string`-typed value is verbatim and unquoted; everything else is JSON. With no schema, scalars best-effort JSON-coerce and fall back to string. This is the single most error-prone rule (identical to GLM-4.5's unquoted strings): emitting `"San Francisco"` for a string parameter yields the literal value *including the quote characters*.
- **Quotes are delimiters, not types.** `y="4"` and `y=4` both coerce to the number `4` under loose/no schema. Quoting an attribute never makes it a string; only a `string` schema type does.
- **Arrays = repetition; single-element arrays need the schema.** Two same-named siblings is unambiguously an array. One occurrence is a scalar under the count heuristic and an array only because the schema says so — a parser without the schema cannot tell `<list>x</list>` (scalar) from a one-element array.
- **Verbatim bodies are delimited by the named closer.** The inline body and any `string`-typed element body are captured up to their matching `</call:NAME>` / `</KEY>`. A body that contains that exact closing sequence truncates early; there is no escaping mechanism. The inline body's risk is minimal because the delimiter includes the tool name (`</call:edit>`), but a short string-typed *element* (e.g. `<note>…</note>`) is more exposed — prefer the inline-body form for any value that might contain markup, or keep such values in the single-string inline payload.
- **Element form vs inline body.** A block call whose body's first non-whitespace content is a child tag matching a known parameter is parsed as element form; otherwise (all-string tool) it is the inline body. A string value that legitimately *starts* with a `<param>`-looking token is the one ambiguity — emit such a tool in element form, or rely on the schema (a tool with structured params is never inline-eligible).
- **No ids; order is the contract.** Calls carry no id; results MUST be returned in call order. Reordering results silently misattributes them.
- **Lenient, not strict XML.** Do not feed pi-native to an XML parser: tag names contain a colon (`call:read`), attribute values may be unquoted, bodies are not entity-encoded, and the stream need not be balanced beyond each call's own open/close. Match the tags as literals (regex / streaming state machine).
- **Streaming.** A stateful parser emits the tool name as soon as `<call:NAME` closes, then streams attribute/child deltas; for an inline body it streams body text incrementally and holds back any partial trailing `</call:` until it can decide whether it is the closer. Coercion of a scalar can only finalize at the value's close tag (a partial number/boolean is not yet valid JSON).
- **Whitespace.** Element/inline bodies preserve all whitespace except one leading and one trailing newline that delimit the block. Attribute and indentation whitespace between tags is insignificant.
## Sources
pi-native is specified here; it is not derived from a published model template. Its two direct influences are documented in this folder:
- Anthropic attribute XML (`<invoke name="…">` / `<parameter name="…">`, "parsed with regular expressions", not required to be valid XML): [`anthropic.md`](./anthropic.md).
- GLM-4.5 schema-driven, **unquoted** string values vs JSON non-strings, and positional (id-less) call↔result correlation: [`glm-4.5.md`](./glm-4.5.md).
+206
View File
@@ -0,0 +1,206 @@
# Qwen3 tool-calling format (Hermes convention)
Tool-calling convention of Alibaba's **Qwen3** family (`Qwen/Qwen3-*`: dense `0.6B–32B` and MoE `30B-A3B`/`235B-A22B`; same template line as `Qwen2.5-*` and `QwQ-32B`). It is the **Hermes** convention — the XML+JSON format originated by NousResearch's Hermes 2 Pro and adopted verbatim by Qwen, plus a long tail of community fine-tunes. The envelope is **ChatML**: every turn is `<|im_start|>{role}\n{body}<|im_end|>\n`. Available tools are advertised in the system turn inside a `<tools>…</tools>` block (one JSON spec per line); the model emits each call as a `<tool_call>\n{json}\n</tool_call>` block whose `arguments` is a **nested JSON object** (not a stringified JSON); tool results are fed back inside `<tool_response>…</tool_response>`. Hybrid reasoning is carried in `<think>…</think>`. The format ships in the model's own `chat_template`, so an inference server enables it with no extra template: vLLM uses `--enable-auto-tool-choice --tool-call-parser hermes` (pair with `--reasoning-parser deepseek_r1` for the thinking split); SGLang exposes the matching parsers (e.g. `--reasoning-parser qwen3`).
Verified against: Qwen's canonical function-calling guide (`qwen.readthedocs.io/en/latest/framework/function_call.html`, read in full incl. the Qwen-Agent + vLLM sections), the byte-exact `chat_template` field of `Qwen/Qwen3-8B`'s `tokenizer_config.json` (HF resolve-cache commit `b968826d9c46dd6066d109eabc6255188de91218`, rendered locally with Jinja2 for the raw streams below) and its `added_tokens_decoder` for token IDs, the NousResearch `Hermes-Function-Calling` README, and the vLLM tool-calling docs (`hermes` parser + Qwen models section).
## Special tokens
Only the three ChatML markers are "special" control tokens (`special=true`, skipped by `skip_special_tokens`). The reasoning and tool markers are also single vocabulary tokens (one ID each) but are registered with `special=false`, i.e. they render as ordinary text and are **not** stripped by `skip_special_tokens`. The `<tools>`/`</tools>` wrapper has **no** dedicated token at all — it is plain text that BPE-splits into several tokens. IDs are from `Qwen/Qwen3-8B` `added_tokens_decoder`.
| Token (verbatim) | ID | `special` | Purpose |
|---|---|---|---|
| `<\|im_start\|>` | 151644 | true | Start of a turn; followed immediately by the role name + `\n` |
| `<\|im_end\|>` | 151645 | true | End of a turn; the chat stop token |
| `<\|endoftext\|>` | 151643 | true | Base EOS / pad token |
| `<think>` | 151667 | false | Opens the reasoning block |
| `</think>` | 151668 | false | Closes the reasoning block |
| `<tool_call>` | 151657 | false | Opens one tool call |
| `</tool_call>` | 151658 | false | Closes one tool call |
| `<tool_response>` | 151665 | false | Opens one tool result |
| `</tool_response>` | 151666 | false | Closes one tool result |
| `<tools>` … `</tools>` | — | — | Plain text wrapper around the tool list in the system turn (not a single token) |
Notes on exactness:
- All markers use the ASCII pipe `|` (U+007C) and ASCII angle brackets. Qwen3 has **no** fullwidth (`|` U+FF5C) or `▁` (U+2581) variants — that is DeepSeek/SentencePiece territory, not Qwen.
- `<|im_start|>` and `<|im_end|>` are the only tokens that matter for splitting turns. Because `<tool_call>`, `</tool_call>`, `<tool_response>`, `<think>`, `</think>` are `special=false`, they survive a `skip_special_tokens=True` decode, which is exactly why the regex-based `hermes` parser can recover them from decoded text.
- The model card confirms `</think>` = token `151668` (used by the reference parsing snippet `output_ids[::-1].index(151668)`).
## Roles / channels / turn structure
ChatML. Each message renders as:
```text
<|im_start|>{role}
{body}<|im_end|>
```
- Roles: `system`, `user`, `assistant`, `tool`. There is no separate "channel" concept; the only sub-stream is the `<think>` reasoning block inside an assistant turn.
- `<|im_end|>\n` terminates every turn. With `add_generation_prompt=True` the prompt ends with `<|im_start|>assistant\n` and the model continues from there.
- **System turn:** if the caller supplies a `system` message it becomes the first turn. When `tools` are present, the tool advertisement is merged **into** that same system turn (the user's system text first, then `\n\n`, then the `# Tools` block — see below). Qwen3 injects no default system prompt when none is given.
- **Tool-result turns use the `user` envelope.** Qwen3's template maps every `role: "tool"` message into a `<|im_start|>user` turn carrying `<tool_response>` blocks (consecutive tool messages are coalesced into one user turn). This differs from classic Hermes 2 Pro, which used a dedicated `<|im_start|>tool` turn for results — Qwen folds them into `user`.
- **Thinking/reasoning:** carried in `<think>…</think>` at the start of an assistant turn (see the Parsing notes for the toggle and the history-rerender rule).
## Tool definitions
Tools are advertised inside the system turn. The template emits a fixed preamble, then each tool object serialized with `tool | tojson` (`json.dumps(..., ensure_ascii=False)`) on **its own line**, then a fixed trailer. Each list element is the full OpenAI tool object `{"type": "function", "function": {...}}` (with a JSON-Schema `parameters` object). The exact, verbatim wrapper Qwen3 produces:
```text
<|im_start|>system
{optional original system content}
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{"type": "function", "function": {"name": "get_current_temperature", "description": "Get current temperature at a location.", "parameters": {"type": "object", "properties": {"location": {"type": "string", "description": "The location to get the temperature for, in the format \"City, State, Country\"."}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"], "description": "The unit to return the temperature in. Defaults to \"celsius\"."}}, "required": ["location"]}}}
{"type": "function", "function": {"name": "get_temperature_date", "description": "Get temperature at a location and date.", "parameters": {"type": "object", "properties": {"location": {"type": "string", "description": "The location to get the temperature for, in the format \"City, State, Country\"."}, "date": {"type": "string", "description": "The date to get the temperature for, in the format \"Year-Month-Day\"."}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"], "description": "The unit to return the temperature in. Defaults to \"celsius\"."}}, "required": ["location", "date"]}}}
</tools>
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
<tool_call>
{"name": <function-name>, "arguments": <args-json-object>}
</tool_call><|im_end|>
```
- If the first message is a `system` message, its content is placed before `# Tools` (separated by a blank line); otherwise the turn opens straight into `# Tools`.
- The trailing instruction is a literal part of the prompt, including the placeholder line `{"name": <function-name>, "arguments": <args-json-object>}` (those angle-bracket tokens are instructions, not emitted output).
- Version note: the original Hermes 2 Pro system prompt additionally embedded a `FunctionCall` pydantic schema line (`{"title": "FunctionCall", "type": "object", "properties": {"name": …, "arguments": …}}`). Qwen3 dropped that line; the wrapper above is exactly what Qwen3 emits.
## Tool-call format
The model emits each call as a `<tool_call>` line, a single-line JSON object, then `</tool_call>`. Minimal single call:
```text
<tool_call>
{"name": "get_current_temperature", "arguments": {"location": "San Francisco, CA, USA", "unit": "celsius"}}
</tool_call>
```
- `arguments` is a **nested JSON object**, not a JSON-encoded string. On the wire it is `"arguments": {"location": "..."}` — never `"arguments": "{\"location\": ...}"`. (The template renders a dict argument via `tojson`; only if a caller stored `arguments` as a pre-serialized string does it pass through verbatim.)
- The call object has exactly two keys, `name` (string) and `arguments` (object). There is no per-call ID on the wire — the OpenAI-style `tool_call_id` is minted by the server, not the model (see API mapping).
- A tool-calling assistant turn may also contain natural-language `content` before the first `<tool_call>`; the template inserts a `\n` between that content and the first call.
## Multiple / parallel tool calls
Parallel calls are emitted as consecutive `<tool_call>…</tool_call>` blocks within a single assistant turn, each separated by a newline:
```text
<|im_start|>assistant
<tool_call>
{"name": "get_current_temperature", "arguments": {"location": "San Francisco, CA, USA"}}
</tool_call>
<tool_call>
{"name": "get_temperature_date", "arguments": {"location": "San Francisco, CA, USA", "date": "2024-10-01"}}
</tool_call><|im_end|>
```
The parser returns these as `tool_calls[0]`, `tool_calls[1]`, … in emission order. The application must execute them and return one `<tool_response>` per call, in the same order.
## Tool-result format
Each executed result is wrapped in `<tool_response>…</tool_response>`. Qwen3 places them inside a **`user`** turn, and **coalesces** consecutive tool results into one turn (one `<tool_response>` block per result, newline-separated, a single closing `<|im_end|>`):
```text
<|im_start|>user
<tool_response>
{"temperature": 26.1, "location": "San Francisco, CA, USA", "unit": "celsius"}
</tool_response>
<tool_response>
{"temperature": 25.9, "location": "San Francisco, CA, USA", "date": "2024-10-01", "unit": "celsius"}
</tool_response><|im_end|>
```
- The body between the tags is the tool's return value (typically a JSON string, but any text is allowed). The function name is **not** repeated inside Qwen3's `<tool_response>` — ordering ties results to calls. (Classic Hermes 2 Pro instead nested `{"name": ..., "content": ...}` inside `<tool_response>` under a `tool` turn; Qwen3's template emits the bare content under a `user` turn.)
- At the OpenAI API layer a result message is `{"role": "tool", "content": "...", "tool_call_id": "..."}`; the template renders only its `content` into a `<tool_response>` block.
## End-to-end example
Complete multi-turn weather exchange in **non-thinking mode** (`enable_thinking=False`), exactly as `apply_chat_template` renders it for the live flow. With thinking disabled, each generation step injects an empty `<think>\n\n</think>\n\n` after `<|im_start|>assistant\n`; the model then emits its tool call / final answer. Copy-pasteable, byte-exact:
```text
<|im_start|>system
You are a helpful assistant. Current Date: 2024-09-30.
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{"type": "function", "function": {"name": "get_current_temperature", "description": "Get current temperature at a location.", "parameters": {"type": "object", "properties": {"location": {"type": "string", "description": "The location to get the temperature for, in the format \"City, State, Country\"."}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"], "description": "The unit to return the temperature in. Defaults to \"celsius\"."}}, "required": ["location"]}}}
</tools>
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
<tool_call>
{"name": <function-name>, "arguments": <args-json-object>}
</tool_call><|im_end|>
<|im_start|>user
What's the temperature in San Francisco now?<|im_end|>
<|im_start|>assistant
<think>
</think>
<tool_call>
{"name": "get_current_temperature", "arguments": {"location": "San Francisco, CA, USA", "unit": "celsius"}}
</tool_call><|im_end|>
<|im_start|>user
<tool_response>
{"temperature": 26.1, "location": "San Francisco, CA, USA", "unit": "celsius"}
</tool_response><|im_end|>
<|im_start|>assistant
<think>
</think>
The current temperature in San Francisco is 26.1°C.<|im_end|>
```
In **thinking mode** (`enable_thinking=True`, the default) the generation prompt instead ends with a bare `<|im_start|>assistant\n` and the model itself produces the `<think>…real reasoning…</think>` block before the `<tool_call>`. (When re-rendering stored history, the template keeps the `<think>` block only for the last assistant message or messages that carry `reasoning_content`, and strips reasoning from earlier turns — see Parsing notes.)
## OpenAI-compatible API mapping
With `--enable-auto-tool-choice --tool-call-parser hermes`, vLLM converts the raw stream into a standard Chat Completions response:
- `finish_reason`: `"tool_calls"` when the turn ended on tool calls (otherwise `"stop"`).
- `message.role`: `"assistant"`; `message.content`: `null` for a pure tool-call turn (any pre-call prose becomes `content`).
- `message.tool_calls[]`: one entry per `<tool_call>` block, each:
- `id`: server-generated, e.g. `"chatcmpl-tool-924d705adb044ff88e0ef3afdd155f15"` (the model emits no ID).
- `type`: `"function"`.
- `function.name`: the call's `name`.
- `function.arguments`: a **JSON string** at the API boundary, e.g. `'{"location": "San Francisco, CA, USA"}'`. The wire format is a nested object, but the server re-serializes it to a string here (`json.loads(...)` it before use), matching OpenAI and Qwen-Agent.
- With thinking + `--reasoning-parser deepseek_r1`, the `<think>…</think>` content is split out into `message.reasoning_content` and removed from `content`.
- Feeding results back: append `{"role": "tool", "content": <result>, "tool_call_id": <id-from-the-call>}` for each result. `tool_call_id` links a result to its call (Qwen3's template ignores the id when rendering — ordering is what reaches the model — but the API still requires it).
Example assistant message returned for the two-call query:
```text
finish_reason='tool_calls'
message.content = None
message.tool_calls = [
{id:'chatcmpl-tool-924d…', type:'function', function:{name:'get_current_temperature', arguments:'{"location": "San Francisco, CA, USA"}'}},
{id:'chatcmpl-tool-7e30…', type:'function', function:{name:'get_temperature_date', arguments:'{"location": "San Francisco, CA, USA", "date": "2024-10-01"}'}},
]
```
## Parsing notes & gotchas
- **Arguments object vs string:** on the wire `arguments` is a nested JSON object; the OpenAI layer hands it back as a JSON string. Code that reads the raw stream must parse an object; code that reads the API must `json.loads` the string. Do not double-encode.
- **`<tools>` is not a token.** Only count on `<|im_start|>`/`<|im_end|>` (and the `*tool_call*`/`*tool_response*`/`*think*` single tokens) being atomic. `<tools>`/`</tools>` are plain text.
- **Regex/streaming parse:** the vLLM `hermes` parser (`vllm/tool_parsers/hermes_tool_parser.py`, `Hermes2ProToolParser`) keys on the literal `<tool_call>` / `</tool_call>` substrings and JSON-decodes the body, supporting multiple blocks per turn. In streaming it buffers from `<tool_call>` until it can incrementally parse `name` then `arguments`; partial argument JSON is emitted as argument deltas. Text before the first `<tool_call>` is streamed as ordinary content.
- **Thinking toggle:** `enable_thinking=False` (passed via `chat_template_kwargs={"enable_thinking": False}` over the OpenAI API, or `tokenizer.apply_chat_template(..., enable_thinking=False)`) injects an empty `<think>\n\n</think>\n\n` into the generation prompt, hard-suppressing reasoning. Soft switches `/think` and `/no_think` in a user/system message flip it per-turn when thinking is enabled. Greedy decoding is discouraged for Qwen3 (repetition risk).
- **History rerender asymmetry:** when `apply_chat_template` re-renders a stored conversation, it emits the `<think>` block only for the final assistant message or messages carrying `reasoning_content`; reasoning from earlier turns is dropped. So a stored intermediate tool-call assistant turn shows no `<think>` block, while the live generation step that produced it was prefixed with one (in non-thinking mode). Reasoning is preserved only within the current multi-step tool sequence (after the last real user query).
- **Reasoning models + stopword templates:** Qwen warns against ReAct-style stopword tool templates for Qwen3, since reasoning text may contain the stopwords and corrupt parsing — use this native Hermes template instead.
- **Robustness:** the format is prompt/template-driven, so malformed output is possible (truncated JSON, missing `</tool_call>`, prose mixed into a call, an array serialized as a string). Production parsers should tolerate and, on failure, fall back to treating the text as content. Named / `required` tool_choice routes through vLLM's structured-outputs backend for guaranteed-parseable arguments.
- **Version/scope:** this `hermes` template covers `Qwen3-*`, `Qwen2.5-*`, and `QwQ-32B`. It does **not** cover `Qwen3-Coder`, which uses a different XML scheme parsed by vLLM's `qwen3_xml` parser — a separate convention.
## Sources
- Qwen function-calling guide: https://qwen.readthedocs.io/en/latest/framework/function_call.html
- Qwen3-8B chat template + token IDs (`tokenizer_config.json`, `chat_template` + `added_tokens_decoder`): https://huggingface.co/Qwen/Qwen3-8B/resolve/main/tokenizer_config.json (verified via HF resolve-cache commit `b968826d9c46dd6066d109eabc6255188de91218`)
- Qwen3-8B model card (thinking modes, `enable_thinking`, `</think>`=151668): https://huggingface.co/Qwen/Qwen3-8B
- NousResearch Hermes-Function-Calling (origin of the convention): https://github.com/NousResearch/Hermes-Function-Calling
- vLLM tool-calling docs (`hermes` parser, Qwen models, auto tool choice): https://docs.vllm.ai/en/latest/features/tool_calling/
+44 -45
View File
@@ -8,7 +8,7 @@
- Key collaborators:
- `packages/coding-agent/src/utils/edit-mode.ts` — selects active edit mode
- `packages/hashline/src/grammar.lark` — canonical constrained-decoding grammar
- `packages/hashline/src/format.ts` — sigils and header constants (`[`, `]`, `#`, `+`, `replace`, `delete`, `insert`)
- `packages/hashline/src/format.ts` — sigils and header constants (`[`, `]`, `#`, `+`, `SWAP`, `DEL`, `INS`)
- `packages/hashline/src/input.ts` — parses `[PATH#TAG]` sections
- `packages/hashline/src/tokenizer.ts` / `packages/hashline/src/parser.ts` — tokenizes and parses ops
- `packages/hashline/src/apply.ts` — applies parsed edits to file text
@@ -28,19 +28,19 @@ Patch language inside `input`:
- **File header**: `[PATH#TAG]`. `TAG` is four uppercase-hex chars — a content-derived hash of the whole normalized file (`computeFileHash()`), recorded in the session snapshot store.
- **Operations**:
- `replace N..M:` — replace original lines N..M with the body rows below.
- `replace block N:` — replace the whole tree-sitter block beginning on line N (its header line through its closing line) with the body rows. The line span is resolved at apply time from the file's parse tree; point N at the line that opens the construct. The resolved span is exactly the node that begins on line N — a leading decorator, attribute, or doc-comment is a separate node and is not included; point N at the first decorator line (Python wraps `@dec` + `def` as one block) or fall back to `replace N..M:` to take a leading line-comment that parses as its own node (e.g. Rust `///`). On success the result echoes the matched span (`replace block N → resolved lines A-B`). Errors (and steers to `replace N..M:`) when the language is unsupported, line N is blank or a closing delimiter, no node begins there, or the resolved block has a syntax error.
- `delete N..M` — delete original lines N..M. No body.
- `delete block N` — delete the whole tree-sitter block beginning on line N (resolved like `replace block N`, with the same decorator/comment caveat). No body. On success the result echoes the matched span (`delete block N → resolved lines A-B`). Same resolution failure modes and `delete N..M` fallback.
- `insert before N:` — insert body rows immediately before line N.
- `insert after N:` — insert body rows immediately after line N.
- `insert after block N:` — insert body rows after the last line of the tree-sitter block beginning on line N. Point N at the line that opens the construct, never its closing delimiter / last visible line; if you can see the last line already, use plain `insert after M:`. Same resolution failure modes and `insert after M:` fallback.
- `insert head:` — insert body rows at the start of the file.
- `insert tail:` — insert body rows at the end of the file.
- `SWAP N.=M:` — replace original lines N.=M with the body rows below.
- `SWAP.BLK N:` — replace the whole tree-sitter block beginning on line N (its header line through its closing line) with the body rows. The line span is resolved at apply time from the file's parse tree; point N at the line that opens the construct. The resolved span is exactly the node that begins on line N — a leading decorator, attribute, or doc-comment is a separate node and is not included; point N at the first decorator line (Python wraps `@dec` + `def` as one block) or fall back to `SWAP N.=M:` to take a leading line-comment that parses as its own node (e.g. Rust `///`). On success the result echoes the matched span (`SWAP.BLK N → resolved lines A-B`). Errors (and steers to `SWAP N.=M:`) when the language is unsupported, line N is blank or a closing delimiter, no node begins there, or the resolved block has a syntax error.
- `DEL N.=M` — delete original lines N.=M. No body.
- `DEL.BLK N` — delete the whole tree-sitter block beginning on line N (resolved like `SWAP.BLK N`, with the same decorator/comment caveat). No body. On success the result echoes the matched span (`DEL.BLK N → resolved lines A-B`). Same resolution failure modes and `DEL N.=M` fallback.
- `INS.PRE N:` — insert body rows immediately before line N.
- `INS.POST N:` — insert body rows immediately after line N.
- `INS.BLK.POST N:` — insert body rows after the last line of the tree-sitter block beginning on line N. Point N at the line that opens the construct, never its closing delimiter / last visible line; if you can see the last line already, use plain `INS.POST M:`. Same resolution failure modes and `INS.POST M:` fallback.
- `INS.HEAD:` — insert body rows at the start of the file.
- `INS.TAIL:` — insert body rows at the end of the file.
- **Body rows**:
- Only body-bearing headers end in `:`.
- Every body row is `+TEXT`; `+` alone adds a blank line.
- `delete` never has body rows.
- `DEL` never has body rows.
- There is no repeat row kind. To keep a line, leave it out of every range; split edits into multiple hunks when needed.
- `-` rows are invalid. Literal text beginning with `-` or `+` must be written as `+-text` / `++text`.
@@ -50,25 +50,25 @@ Anchors come from `read`/`search` output. `read` emits a `[PATH#TAG]` header fro
The canonical grammar is strict, but the hand parser accepts a few non-dangerous variants:
- `replace N:` — accepted as `replace N..N:`.
- `delete N` — accepted as single-line delete.
- Missing trailing colon on `replace` or `insert` — accepted.
- `replace N-M:`, `replace N…M:`, and `replace N M:` — accepted as `replace N..M:`.
- `SWAP N:` — accepted as `SWAP N.=N:`.
- `DEL N` — accepted as single-line delete.
- Missing trailing colon on `SWAP` or `INS` — accepted.
- `SWAP N-M:`, `SWAP N…M:`, `SWAP N M:`, and legacy `SWAP N.=M:` — accepted as `SWAP N.=M:`.
- Bare body rows with no `+` prefix are auto-prepended with `+` and a `BARE_BODY_AUTO_PIPED_WARNING` is appended.
- `*** Begin Patch` / `*** End Patch` envelopes are silently consumed. `*** Abort` terminates parsing silently — ops parsed before the marker still apply, no warning surfaced.
- Some malformed bracketed headers are recovered after stripping apply-patch path noise such as `Update File:` / `Add File:` and extra `***`, but the recovered header still needs a valid four-hex tag for the patcher to apply it.
- `*** Update File:` / `*** Add File:` / `*** Delete File:` / `*** Move to:` apply_patch sentinels inside the diff body throw an `apply_patch sentinel … is not valid in hashline` error.
- `@@`-bracketed hunk headers are rejected with guidance to write a verb header.
- Bare `N` and bare `N M` / `N..M` headers are rejected with guidance to write `replace` or `delete`.
- `delete N..M:` and any body rows under `delete` / `delete block` are rejected.
- Empty `replace` / `insert` / `replace block` hunks are rejected.
- Bare `N` and bare `N M` / `N.=M` headers are rejected with guidance to write `SWAP` or `DEL`.
- `DEL N.=M:` and any body rows under `DEL` / `DEL.BLK` are rejected.
- Empty `SWAP` / `INS` / `SWAP.BLK` hunks are rejected.
- `-` body rows are rejected with `MINUS_ROW_REJECTED`.
- `replace block N:` / `delete block N` / `insert after block N:` require a wired tree-sitter resolver; `replace block` and `insert after block` additionally need at least one `+TEXT` body row, while `delete block` takes none. An unresolvable block (unsupported language, blank/closing-delimiter line, no node beginning on N, or a syntax error in the resolved block) is rejected on the apply/final-preview path; the streaming preview silently drops it instead. Exception: `insert after block N:` anchored on a pure closing-delimiter line is lowered to plain `insert after N:` with a warning — line N is the end of a block, and inserting after that end is exactly what the plain form does.
- `SWAP.BLK N:` / `DEL.BLK N` / `INS.BLK.POST N:` require a wired tree-sitter resolver; `SWAP.BLK` and `INS.BLK.POST` additionally need at least one `+TEXT` body row, while `DEL.BLK` takes none. An unresolvable block (unsupported language, blank/closing-delimiter line, no node beginning on N, or a syntax error in the resolved block) is rejected on the apply/final-preview path; the streaming preview silently drops it instead. Exception: `INS.BLK.POST N:` anchored on a pure closing-delimiter line is lowered to plain `INS.POST N:` with a warning — line N is the end of a block, and inserting after that end is exactly what the plain form does.
## Outputs
- Single-shot tool result; hashline mode does not use a `resolve` preview/apply handshake.
- `content` contains one text block per call. For a successful single-file edit it is the post-edit `[path#TAG]` section header (a fresh snapshot tag for the written content), followed by a compact diff preview from `packages/hashline/src/diff-preview.ts` when one is emitted.
- When the patch used `replace block`/`delete block`/`insert after block` ops (and the apply matched the tagged content), one `replace block N → resolved lines A-B (K lines)` line per block op (single-line spans render `resolved line A (1 line)`; insert-after appends `; body lands after line B`) is inserted between the `[PATH#TAG]` header and the diff preview, so the caller can confirm tree-sitter resolved the construct it intended.
- When the patch used `SWAP.BLK`/`DEL.BLK`/`INS.BLK.POST` ops (and the apply matched the tagged content), one `SWAP.BLK N → resolved lines A-B (K lines)` line per block op (single-line spans render `resolved line A (1 line)`; INS.BLK.POST appends `; body lands after line B`) is inserted between the `[PATH#TAG]` header and the diff preview, so the caller can confirm tree-sitter resolved the construct it intended.
- Parse, apply, or recovery warnings are appended as:
```text
@@ -103,7 +103,7 @@ Replace line 1 with two lines:
```text
[a.ts#0A3B]
replace 1..1:
SWAP 1.=1:
+const X = "b";
+export const Y = X;
```
@@ -112,7 +112,7 @@ Insert below line 5:
```text
[a.ts#0A3B]
insert after 5:
INS.POST 5:
+console.log(X + Y);
```
@@ -120,42 +120,41 @@ Insert above line 5:
```text
[a.ts#0A3B]
insert before 5:
INS.PRE 5:
+console.log(X + Y);
```
Delete lines 4..5 entirely:
Delete lines 4.=5 entirely:
```text
[a.ts#0A3B]
delete 4..5
DEL 4.=5
```
Insert at start and end of file:
```text
[a.ts#0A3B]
insert head:
INS.HEAD:
+// header
insert tail:
INS.TAIL:
+// trailer
```
Multi-file:
```text
[src/a.ts#0A3B]
replace 4..4:
SWAP 4.=4:
+const enabled = true;
[src/b.ts#1F7C]
delete 20
DEL 20
```
## Limits & Caps
- File snapshot tags are exactly four uppercase-hex chars — content-derived hashes (`computeFileHash()`) recorded in the per-session snapshot store.
- The visible mismatch report shows 2 lines of context on each side (`MISMATCH_CONTEXT`) in `packages/hashline/src/messages.ts`.
- Stale-anchor recovery uses `fuzzFactor: 0` in `packages/hashline/src/recovery.ts`.
- `HL_FILE_PREFIX` is `[`, `HL_FILE_SUFFIX` is `]`, `HL_PAYLOAD_REPLACE` is `+`, `HL_RANGE_SEP` is `..`, `HL_FILE_HASH_SEP` is `#`, and hunk keyword constants are `replace` / `delete` / `insert` (`packages/hashline/src/format.ts`).
- `HL_FILE_PREFIX` is `[`, `HL_FILE_SUFFIX` is `]`, `HL_PAYLOAD_REPLACE` is `+`, `HL_RANGE_SEP` is `.=`, `HL_FILE_HASH_SEP` is `#`, and hunk keyword constants are `SWAP` / `DEL` / `INS` (`packages/hashline/src/format.ts`).
## Errors
- Missing section header:
@@ -163,29 +162,29 @@ delete 20
- Missing tag for any section:
- `Missing hashline snapshot tag for edit to <path>; use \`[<path>#tag]\` from your latest read/search output. To create a new file, use the write tool.`
- Stray payload line:
- `line N: payload line has no preceding hunk header. Use \`replace N..M:\`, \`delete N..M\`, or \`insert before|after|head|tail:\` above the body. Got "...".`
- `line N: payload line has no preceding hunk header. Use \`SWAP N.=M:\`, \`DEL N.=M\`, or \`INS.PRE|POST|HEAD|TAIL:\` above the body. Got "...".`
- Minus row:
- ``line N: `-` rows are not valid; hashline ranges already name the lines being changed. To insert a literal line starting with `-`, write `+-…`.``
- Empty body-bearing hunk:
- `line N: \`replace N..M:\` needs at least one \`+TEXT\` body row. To delete lines, use \`delete N..M\`.`
- `line N: \`insert\` needs at least one \`+TEXT\` body row.`
- `line N: \`replace block N:\` needs at least one \`+TEXT\` body row. To delete a block, use \`delete N..M\` with the block's line range.`
- `line N: \`SWAP N.=M:\` needs at least one \`+TEXT\` body row. To delete lines, use \`DEL N.=M\`.`
- `line N: \`INS\` needs at least one \`+TEXT\` body row.`
- `line N: \`SWAP.BLK N:\` needs at least one \`+TEXT\` body row. To delete a block, use \`DEL.BLK N\`.`
- Unresolvable block anchor (apply / final-preview path only):
- `line N: \`replace block X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`replace X..M:\` with the block's explicit end line instead.` — followed by a blank line and numbered `*`-marked context rows around line X (same shape as the mismatch preview).
- `line N: \`insert after block X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`insert after M:\` with the block's explicit last line instead.` — same context preview.
- `line N: \`SWAP.BLK X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`SWAP X.=M:\` with the block's explicit end line instead.` — followed by a blank line and numbered `*`-marked context rows around line X (same shape as the mismatch preview).
- `line N: \`INS.BLK.POST X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`INS.POST M:\` with the block's explicit last line instead.` — same context preview.
- Delete with body:
- `line N: \`delete N..M\` does not take body rows. Remove the body, or use \`replace N..M:\`.`
- `line N: \`delete block N\` does not take body rows. Remove the body, or use \`replace block N:\` to replace the block.`
- `line N: \`DEL N.=M\` does not take body rows. Remove the body, or use \`SWAP N.=M:\`.`
- `line N: \`DEL.BLK N\` does not take body rows. Remove the body, or use \`SWAP.BLK N:\` to replace the block.`
- Range out of order:
- `line N: range A..B ends before it starts.`
- `line N: range A.=B ends before it starts.`
- Overlapping hunks on the same anchor:
- `line N: anchor line X is already targeted by another hunk on line Y. Issue ONE hunk per range; payload is only the final desired content, never a before/after pair.`
- apply_patch / unified-diff contamination:
- `line N: apply_patch sentinel "*** …" is not valid in hashline. File sections start with \`[path#HASH]\` (no \`Update File:\` / \`Add File:\` keyword). Use \`replace N..M:\`, \`delete N..M\`, or \`insert before|after|head|tail:\` ops.`
- `line N: unified-diff hunk header (\`@@ -N,M +N,M @@\`) is not valid in hashline. Use \`replace N..M:\`, \`delete N..M\`, or \`insert before|after|head|tail:\` ops.`
- `line N: \`@@\`-bracketed hunk header "@@ …" is not valid in hashline. Drop the \`@@ ... @@\` brackets and write a verb header such as \`replace N..M:\`.`
- `line N: hunk headers need a verb. Use \`replace N..N:\` to replace, or \`delete N\` to delete.`
- `line N: bare range hunk header "N M" is not valid. Hunk headers need a verb: write \`replace N..M:\` or \`delete N..M\`.`
- `line N: apply_patch sentinel "*** …" is not valid in hashline. File sections start with \`[path#HASH]\` (no \`Update File:\` / \`Add File:\` keyword). Use \`SWAP N.=M:\`, \`DEL N.=M\`, or \`INS.PRE|POST|HEAD|TAIL:\` ops.`
- `line N: unified-diff hunk header (\`@@ -N,M +N,M @@\`) is not valid in hashline. Use \`SWAP N.=M:\`, \`DEL N.=M\`, or \`INS.PRE|POST|HEAD|TAIL:\` ops.`
- `line N: \`@@\`-bracketed hunk header "@@ …" is not valid in hashline. Drop the \`@@ ... @@\` brackets and write a verb header such as \`SWAP N.=M:\`.`
- `line N: hunk headers need a verb. Use \`SWAP N.=N:\` to replace, or \`DEL N\` to delete.`
- `line N: bare range hunk header "N M" is not valid. Hunk headers need a verb: write \`SWAP ${bareRange[1]}.=${bareRange[2]}:\` or \`DEL ${bareRange[1]}.=${bareRange[2]}\`.`
- Out-of-range anchor:
- `Line N does not exist (file has M lines)`
- Stale snapshot tag: the `Patcher` first attempts snapshot-based recovery. When recovery cannot prove a valid result it throws `MismatchError`, which distinguishes recognized-but-drifted hashes from never-recorded hashes. The error includes the current file hash plus context around each anchor.
+13 -13
View File
@@ -24,18 +24,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "15.13.1",
"@oh-my-pi/omp-stats": "15.13.1",
"@oh-my-pi/pi-agent-core": "15.13.1",
"@oh-my-pi/pi-ai": "15.13.1",
"@oh-my-pi/pi-catalog": "15.13.1",
"@oh-my-pi/pi-coding-agent": "15.13.1",
"@oh-my-pi/pi-mnemopi": "15.13.1",
"@oh-my-pi/pi-natives": "15.13.1",
"@oh-my-pi/pi-tui": "15.13.1",
"@oh-my-pi/pi-utils": "15.13.1",
"@oh-my-pi/pi-wire": "15.13.1",
"@oh-my-pi/snapcompact": "15.13.1",
"@oh-my-pi/hashline": "15.13.3",
"@oh-my-pi/omp-stats": "15.13.3",
"@oh-my-pi/pi-agent-core": "15.13.3",
"@oh-my-pi/pi-ai": "15.13.3",
"@oh-my-pi/pi-catalog": "15.13.3",
"@oh-my-pi/pi-coding-agent": "15.13.3",
"@oh-my-pi/pi-mnemopi": "15.13.3",
"@oh-my-pi/pi-natives": "15.13.3",
"@oh-my-pi/pi-tui": "15.13.3",
"@oh-my-pi/pi-utils": "15.13.3",
"@oh-my-pi/pi-wire": "15.13.3",
"@oh-my-pi/snapcompact": "15.13.3",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -118,7 +118,7 @@
"fmt:ts": "bun run fmt:tools && bun run --workspaces --if-present fmt",
"fmt:tools": "biome format --write . --no-errors-on-unmatched",
"fmt:rs": "bun scripts/run-rs-task.ts fmt:rs",
"fix": "bun run --parallel fix:ts fix:rs fix:changelogs",
"fix": "bun run --parallel fix:ts fix:rs",
"fix:all": "bun run --parallel fix:ts:all fix:rs fix:changelogs",
"fix:ts": "bun run fix:tools && bun run --workspaces --if-present fix",
"fix:ts:all": "bun run fix:tools:all && bun run --workspaces --if-present fix",
+35
View File
@@ -2,6 +2,41 @@
## [Unreleased]
## [15.13.3] - 2026-06-15
### Added
- Added the `interruptible` tool field: when set, the agent loop may abort the tool mid-execution to deliver a queued steering message (honored only in `immediate` interrupt mode).
- Added support for `gemini` and `gemma` as valid owned tool syntax values in environment configuration
### Fixed
- Fixed `pruneToolOutputs` blanking tiny tool results during overflow pruning: results below `50` tokens (`MIN_PRUNE_TOKENS`) are no longer replaced with the `[Output truncated - N tokens]` placeholder, which cost more tokens than the result itself and churned the prompt cache for zero savings.
## [15.13.2] - 2026-06-15
### Breaking Changes
- Removed `harmony-leak` exports from the `@oh-my-pi/pi-agent-core` package entrypoint
- Replaced the experimental `promptToolCalls` agent/loop option with `toolCallSyntax`, selecting an explicit in-band tool-call grammar instead of a boolean GLM-only mode.
### Added
- Added support for selecting owned in-band tool-call syntax via `PI_OWNED_TOOLS=<syntax>` (for example `hermes` or `qwen3`) while preserving legacy `PI_OWNED_TOOLS=1/true` as GLM mode
- Added owned in-band tool calling for multiple syntaxes (`glm`, `hermes`, `kimi`, `xml`, `anthropic`, `deepseek`, `harmony`, `pi-native`, `qwen3`). Owned mode sends no native provider tools, appends a syntax-specific prompt/catalog, re-encodes prior tool calls/results as grammar-owned text, and parses streamed model output back into canonical tool calls.
- Added tool-example folding to `normalizeTools`: when given a model's affinity syntax (resolved via `preferredToolSyntax`), it renders each tool's `examples` into an `<examples>` block in that native syntax and appends it to the wire description. Wired through both context paths (fresh build and append-only `takeSnapshot`/`build` via a new `exampleSyntax` build option), with the `_i` intent-field placeholder added to examples when intent tracing injects it.
- Added the `abortOnFabricatedToolResult` option to `AgentOptions`/`AgentLoopConfig` (default `true`): when owned tool calling is active and the model fabricates a tool result mid-turn, `true` aborts the provider request immediately while `false` lets it finish and discards the fabricated continuation.
### Changed
- Added owned in-band syntax support to `Agent` loop configuration resolution by selecting syntax from `toolCallSyntax` or `PI_OWNED_TOOLS` when present
### Fixed
- Fixed append-only context cache fingerprinting to account for `exampleSyntax`, so switching tool-call syntax rebuilds cached prompts with the correct injected tool examples
- Fixed owned in-band tool-calling requests to omit `toolChoice` after stripping native tools, preventing invalid tool-choice requests
- Fixed owned tool calling letting the model fabricate tool results by treating grammar-owned tool-result markers in assistant text as a hard turn boundary: calls before the fabrication are kept, fabricated results and dependent calls are dropped, and the real result is fed back on the next turn.
## [15.13.1] - 2026-06-15
### Added
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "15.13.1",
"version": "15.13.3",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+118 -12
View File
@@ -15,7 +15,13 @@ import {
validateToolArguments,
zodToWireSchema,
} from "@oh-my-pi/pi-ai";
import { logger, sanitizeText } from "@oh-my-pi/pi-utils";
import {
encodeInbandToolHistory,
renderInbandToolPrompt,
renderToolExamples,
type ToolCallSyntax,
wrapInbandToolStream,
} from "@oh-my-pi/pi-ai/grammar";
import {
createHarmonyAuditEvent,
detectHarmonyLeakInAssistantMessage,
@@ -25,7 +31,9 @@ import {
isHarmonyLeakMitigationTarget,
recoverHarmonyToolCall,
signalListLabel,
} from "./harmony-leak";
} from "@oh-my-pi/pi-ai/utils/harmony-leak";
import { preferredToolSyntax } from "@oh-my-pi/pi-catalog/identity";
import { logger, sanitizeText } from "@oh-my-pi/pi-utils";
import { type AgentRunCoverage, type AgentRunSummary, ToolCallBlockedError } from "./run-collector";
import {
type AgentTelemetry,
@@ -66,6 +74,14 @@ const ABORTED: unique symbol = Symbol("agent-loop-aborted");
*/
const MAX_PAUSED_TURN_CONTINUATIONS = 8;
/**
* Cadence (ms) for polling queued steering while an `interruptible` tool is in
* flight, so a steer cuts the wait short instead of sitting idle until the
* tool's own window elapses. A cheap synchronous queue check; latency-bounded
* at one tick.
*/
const STEERING_INTERRUPT_POLL_MS = 250;
class HarmonyLeakInterruption extends Error {
constructor(
readonly detection: HarmonyDetection,
@@ -76,6 +92,27 @@ class HarmonyLeakInterruption extends Error {
this.name = "HarmonyLeakInterruption";
}
}
function resolveOwnedToolSyntaxFromEnv(value: string | undefined): ToolCallSyntax | undefined {
switch (value) {
case "1":
case "true":
return "glm";
case "glm":
case "hermes":
case "kimi":
case "xml":
case "anthropic":
case "deepseek":
case "harmony":
case "pi":
case "qwen3":
case "gemini":
case "gemma":
return value;
default:
return undefined;
}
}
type AssistantContentBlock = AssistantMessage["content"][number];
type AssistantToolCallBlock = Extract<AssistantContentBlock, { type: "toolCall" }>;
@@ -491,7 +528,11 @@ function injectIntentIntoSchema(schema: unknown, mode: "require" | "optional" =
};
}
export function normalizeTools(tools: AgentContext["tools"], injectIntent: boolean): Context["tools"] {
export function normalizeTools(
tools: AgentContext["tools"],
injectIntent: boolean,
exampleSyntax?: ToolCallSyntax,
): Context["tools"] {
injectIntent = injectIntent && Bun.env.PI_NO_INTENT !== "1";
return tools?.map(t => {
const intentMode = resolveIntentMode(t.intent);
@@ -505,7 +546,12 @@ export function normalizeTools(tools: AgentContext["tools"], injectIntent: boole
}
}
const description = t.description ?? "";
return { ...t, parameters, description };
const injectExampleIntent = injectIntent && intentMode !== "omit";
const examplesBlock = exampleSyntax
? renderToolExamples({ ...t, parameters }, exampleSyntax, injectExampleIntent ? INTENT_FIELD : undefined)
: "";
const finalDescription = examplesBlock ? `${description}\n\n${examplesBlock}` : description;
return { ...t, parameters, description: finalDescription };
});
}
@@ -884,18 +930,37 @@ async function streamAssistantResponse(
let llmContext: Context;
if (config.appendOnlyContext) {
config.appendOnlyContext.syncMessages(normalizedMessages);
llmContext = config.appendOnlyContext.build(context, { intentTracing: !!config.intentTracing });
llmContext = config.appendOnlyContext.build(context, {
intentTracing: !!config.intentTracing,
exampleSyntax: preferredToolSyntax(config.model.id),
});
} else {
llmContext = {
systemPrompt: context.systemPrompt,
messages: normalizedMessages,
tools: normalizeTools(context.tools, !!config.intentTracing),
tools: normalizeTools(context.tools, !!config.intentTracing, preferredToolSyntax(config.model.id)),
};
}
if (config.transformProviderContext) {
llmContext = config.transformProviderContext(llmContext, config.model);
}
// Owned tool calling: take tool calls away from the provider and run them
// through the selected in-band prompt syntax. `PI_OWNED_TOOLS=1` still
// force-enables GLM; `PI_OWNED_TOOLS=<syntax>` force-enables that syntax.
const ownedSyntax: ToolCallSyntax | undefined =
config.toolCallSyntax ?? resolveOwnedToolSyntaxFromEnv(Bun.env.PI_OWNED_TOOLS);
let promptToolWireTools: Context["tools"];
if (ownedSyntax && llmContext.tools && llmContext.tools.length > 0) {
promptToolWireTools = llmContext.tools;
llmContext = {
...llmContext,
systemPrompt: [...(llmContext.systemPrompt ?? []), renderInbandToolPrompt(promptToolWireTools, ownedSyntax)],
messages: encodeInbandToolHistory(llmContext.messages, ownedSyntax, promptToolWireTools),
tools: undefined,
};
}
const streamFunction = streamFn || streamSimple;
// Resolve API key (important for expiring tokens) — do this before resolving
@@ -920,12 +985,22 @@ async function streamAssistantResponse(
: harmonyAbortController.signal
: signal;
const repetitionAbortController = new AbortController();
const finalRequestSignal = requestSignal
? AbortSignal.any([requestSignal, repetitionAbortController.signal])
: repetitionAbortController.signal;
// Owned tool calling: aborted by the stream wrapper when the model starts
// fabricating a `<tool_response>`, so the provider stops generating the rest of
// the hallucinated turn. Merged into the provider signal ONLY (not
// `requestSignal`), so it cancels the request without tripping the loop's
// external-abort handling (`abortRacePromise` / `requestSignal.aborted`).
const promptToolAbortController = ownedSyntax ? new AbortController() : undefined;
const providerAbortSignals: AbortSignal[] = [];
if (requestSignal) providerAbortSignals.push(requestSignal);
providerAbortSignals.push(repetitionAbortController.signal);
if (promptToolAbortController) providerAbortSignals.push(promptToolAbortController.signal);
const finalRequestSignal =
providerAbortSignals.length === 1 ? providerAbortSignals[0]! : AbortSignal.any(providerAbortSignals);
const effectiveTemperature =
harmonyRetryAttempt > 0 && config.temperature !== undefined ? config.temperature + 0.05 : config.temperature;
const effectiveToolChoice = dynamicToolChoice ?? config.toolChoice;
// Owned tool calling sends no native tools, so any tool_choice would error.
const effectiveToolChoice = ownedSyntax ? undefined : (dynamicToolChoice ?? config.toolChoice);
const effectiveReasoning = dynamicReasoning ?? config.reasoning;
const effectiveDisableReasoning = dynamicDisableReasoning ?? config.disableReasoning;
@@ -970,7 +1045,7 @@ async function streamAssistantResponse(
try {
return await runInActiveSpan(chatSpan, async () => {
const response = await streamFunction(config.model, llmContext, {
let response = await streamFunction(config.model, llmContext, {
...config,
// Hand streamSimple a resolver so its central auth-retry policy can
// re-resolve on 401 / usage-limit: the initial step reuses the key
@@ -993,6 +1068,20 @@ async function streamAssistantResponse(
signal: finalRequestSignal,
onResponse: captureOnResponse,
});
if (promptToolWireTools && ownedSyntax) {
// Re-materialize in-band tool-call text as native toolCall content blocks
// so the rest of the loop executes them unchanged. When the model starts
// fabricating tool results, the abort callback cancels the provider — unless
// `abortOnFabricatedToolResult` is false, in which case the stream drains and
// the fabricated continuation is discarded without aborting.
response = wrapInbandToolStream(
response,
promptToolWireTools,
ownedSyntax,
() => promptToolAbortController?.abort(),
config.abortOnFabricatedToolResult ?? true,
);
}
let partialMessage: AssistantMessage | null = null;
let addedPartial = false;
@@ -1716,7 +1805,24 @@ async function executeToolCalls(
}
}
await Promise.allSettled(tasks);
// While an interruptible tool is in flight (e.g. a `job` poll blocking on
// background work), a queued steer would otherwise wait out the tool's own
// window. Poll the steering queue and let checkSteering() abort the shared
// tool signal so the wait returns early; the boundary dequeue below then
// injects it. Gated on immediate-interrupt mode + an interruptible tool;
// checkSteering is idempotent (no-op once triggered).
const watchSteeringWhileRunning =
shouldInterruptImmediately &&
(hasSteeringMessages !== undefined || getSteeringMessages !== undefined) &&
records.some(r => r.tool?.interruptible === true);
const steeringWatchTimer = watchSteeringWhileRunning
? setInterval(() => void checkSteering(), STEERING_INTERRUPT_POLL_MS)
: undefined;
try {
await Promise.allSettled(tasks);
} finally {
if (steeringWatchTimer !== undefined) clearInterval(steeringWatchTimer);
}
// Yield after batch tool execution to let GC and I/O catch up,
// especially when tool results are large (e.g. bash output).
await yieldIfDue();
+17 -1
View File
@@ -22,11 +22,12 @@ import {
type ToolChoice,
type ToolResultMessage,
} from "@oh-my-pi/pi-ai";
import type { ToolCallSyntax } from "@oh-my-pi/pi-ai/grammar";
import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { logger } from "@oh-my-pi/pi-utils";
import { abortReasonText, agentLoop, agentLoopContinue } from "./agent-loop";
import type { AppendOnlyContextManager } from "./append-only-context";
import type { HarmonyAuditEvent } from "./harmony-leak";
import type {
AgentContext,
AgentEvent,
@@ -220,6 +221,15 @@ export interface AgentOptions {
/** Enable intent tracing schema injection/stripping in the harness. */
intentTracing?: boolean;
/** Owned tool-calling syntax. Undefined keeps provider-native tool calling. */
toolCallSyntax?: ToolCallSyntax;
/**
* When owned tool calling is active and the model fabricates a tool result
* mid-turn: `true` (default) aborts the provider request immediately; `false`
* drains the request and discards the fabricated continuation. Forwarded to
* the loop's {@link AgentLoopConfig.abortOnFabricatedToolResult}.
*/
abortOnFabricatedToolResult?: boolean;
/** Dynamic tool choice override, resolved per LLM call. */
getToolChoice?: () => ToolChoice | undefined;
@@ -316,6 +326,8 @@ export class Agent {
#preferWebsockets?: boolean;
#transformToolCallArguments?: (args: Record<string, unknown>, toolName: string) => Record<string, unknown>;
#intentTracing: boolean;
#toolCallSyntax?: ToolCallSyntax;
#abortOnFabricatedToolResult?: boolean;
#getToolChoice?: () => ToolChoice | undefined;
#onPayload?: SimpleStreamOptions["onPayload"];
#onResponse?: SimpleStreamOptions["onResponse"];
@@ -378,6 +390,8 @@ export class Agent {
this.#preferWebsockets = opts.preferWebsockets;
this.#transformToolCallArguments = opts.transformToolCallArguments;
this.#intentTracing = opts.intentTracing === true;
this.#toolCallSyntax = opts.toolCallSyntax;
this.#abortOnFabricatedToolResult = opts.abortOnFabricatedToolResult;
this.#getToolChoice = opts.getToolChoice;
this.#onAssistantMessageEvent = opts.onAssistantMessageEvent;
this.#onHarmonyLeak = opts.onHarmonyLeak;
@@ -1023,6 +1037,8 @@ export class Agent {
cursorOnToolResult,
transformToolCallArguments: this.#transformToolCallArguments,
intentTracing: this.#intentTracing,
toolCallSyntax: this.#toolCallSyntax,
abortOnFabricatedToolResult: this.#abortOnFabricatedToolResult,
appendOnlyContext: this.#appendOnlyContext,
beforeToolCall: this.beforeToolCall ? (ctx, signal) => this.beforeToolCall?.(ctx, signal) : undefined,
afterToolCall: this.afterToolCall ? (ctx, signal) => this.afterToolCall?.(ctx, signal) : undefined,
+4 -1
View File
@@ -15,6 +15,7 @@
*/
import type { Context, Message, Tool } from "@oh-my-pi/pi-ai";
import type { ToolCallSyntax } from "@oh-my-pi/pi-ai/grammar";
import { normalizeTools } from "./agent-loop";
import type { AgentContext } from "./types";
@@ -33,6 +34,7 @@ export interface StablePrefixSnapshot {
export interface BuildOptions {
/** Inject the `_i` intent field into tool schemas (must match agent-loop's normalizeTools). */
intentTracing: boolean;
exampleSyntax?: ToolCallSyntax;
}
/**
@@ -268,7 +270,7 @@ export class AppendOnlyContextManager {
function takeSnapshot(context: AgentContext, options: BuildOptions): StablePrefixSnapshot {
const systemPrompt = [...context.systemPrompt];
const tools = normalizeTools(context.tools, options.intentTracing) ?? [];
const tools = normalizeTools(context.tools, options.intentTracing, options.exampleSyntax) ?? [];
return {
systemPrompt,
tools,
@@ -288,6 +290,7 @@ function computeFingerprint(systemPrompt: string[], tools: Tool[], options: Buil
cw: t.customWireName,
})),
i: options.intentTracing,
ex: options.exampleSyntax,
});
let hash = 0;
for (let i = 0; i < payload.length; i++) {
@@ -6,6 +6,7 @@
*/
import type { ApiKey, Model } from "@oh-my-pi/pi-ai";
import { preferredToolSyntax } from "@oh-my-pi/pi-catalog/identity";
import { prompt } from "@oh-my-pi/pi-utils";
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
import type { AgentMessage } from "../types";
@@ -290,7 +291,7 @@ export async function generateBranchSummary(
// Transform to LLM-compatible messages, then serialize to text
// Serialization prevents the model from treating it as a conversation to continue
const llmMessages = (options.convertToLlm ?? defaultConvertToLlm)(messages);
const conversationText = serializeConversation(llmMessages);
const conversationText = serializeConversation(llmMessages, preferredToolSyntax(model.id));
// Build prompt
const instructions = customInstructions || BRANCH_SUMMARY_PROMPT;
+4 -3
View File
@@ -18,6 +18,7 @@ import {
type Usage,
withAuth,
} from "@oh-my-pi/pi-ai";
import { preferredToolSyntax } from "@oh-my-pi/pi-catalog/identity";
import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
import { countTokens } from "@oh-my-pi/pi-natives";
import { logger, prompt } from "@oh-my-pi/pi-utils";
@@ -642,7 +643,7 @@ export async function generateSummary(
// Serialize conversation to text so model doesn't try to continue it
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
const conversationText = serializeConversation(llmMessages);
const conversationText = serializeConversation(llmMessages, preferredToolSyntax(model.id));
// Build the prompt with conversation wrapped in tags
let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
@@ -790,7 +791,7 @@ async function generateShortSummary(
): Promise<string> {
const maxTokens = Math.min(512, Math.floor(0.2 * reserveTokens));
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(recentMessages);
const conversationText = serializeConversation(llmMessages);
const conversationText = serializeConversation(llmMessages, preferredToolSyntax(model.id));
let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
if (historySummary) {
@@ -1155,7 +1156,7 @@ async function generateTurnPrefixSummary(
const maxTokens = Math.floor(0.5 * reserveTokens); // Smaller budget for turn prefix
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(messages);
const conversationText = serializeConversation(llmMessages);
const conversationText = serializeConversation(llmMessages, preferredToolSyntax(model.id));
const promptText = `<conversation>\n${conversationText}\n</conversation>\n\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;
const summarizationMessages = [
{
+12 -1
View File
@@ -81,6 +81,16 @@ function createPrunedNotice(tokens: number): string {
return `[Output truncated - ${tokens} tokens]`;
}
/**
* Generic age-based pruning floor. Below this, blanking a result to
* `[Output truncated - N tokens]` recovers nothing — the placeholder itself
* costs ~8 tokens, so a sub-floor result grows the context (and churns the
* prompt cache) instead of shrinking it. Superseded/useless results keep their
* own rules: useless already drops no-savings candidates, superseded prunes for
* correctness regardless of size.
*/
const MIN_PRUNE_TOKENS = 50;
function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefined {
if (entry.type !== "message") return undefined;
const message = entry.message as AgentMessage;
@@ -271,7 +281,8 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
// any age).
const superseded = supersededMessages?.has(message) ?? false;
const useless = uselessMessages?.has(message) ?? false;
if (!superseded && !useless && (accumulatedTokens < config.protectTokens || isProtected)) {
const tooSmall = tokens < MIN_PRUNE_TOKENS;
if (!superseded && !useless && (accumulatedTokens < config.protectTokens || isProtected || tooSmall)) {
accumulatedTokens += tokens;
continue;
}
+44 -11
View File
@@ -2,7 +2,8 @@
* Shared utilities for compaction and branch summarization.
*/
import type { Message } from "@oh-my-pi/pi-ai";
import type { Message, ToolCall } from "@oh-my-pi/pi-ai";
import { type Grammar, type GrammarToolResult, getInbandGrammar, type ToolCallSyntax } from "@oh-my-pi/pi-ai/grammar";
import { formatGroupedPaths, prompt } from "@oh-my-pi/pi-utils";
import type { AgentMessage } from "../types";
import fileOperationsTemplate from "./prompts/file-operations.md" with { type: "text" };
@@ -188,7 +189,8 @@ function truncateForSummary(text: string, maxChars: number): string {
* This prevents the model from treating it as a conversation to continue.
* Call convertToLlm() first to handle custom message types.
*/
export function serializeConversation(messages: Message[]): string {
export function serializeConversation(messages: Message[], syntax?: ToolCallSyntax): string {
const grammar = syntax ? getInbandGrammar(syntax) : undefined;
const parts: string[] = [];
// Tool results flagged contextually useless (and their paired calls) are
@@ -215,7 +217,7 @@ export function serializeConversation(messages: Message[]): string {
} else if (msg.role === "assistant") {
const textParts: string[] = [];
const thinkingParts: string[] = [];
const toolCalls: string[] = [];
const toolCalls: ToolCall[] = [];
for (const block of msg.content) {
if (block.type === "text") {
@@ -224,22 +226,18 @@ export function serializeConversation(messages: Message[]): string {
thinkingParts.push(block.thinking);
} else if (block.type === "toolCall") {
if (uselessCallIds.has(block.id)) continue;
const args = block.arguments as Record<string, unknown>;
const argsStr = Object.entries(args)
.map(([k, v]) => `${k}=${JSON.stringify(v)}`)
.join(", ");
toolCalls.push(`${block.name}(${argsStr})`);
toolCalls.push(block);
}
}
if (thinkingParts.length > 0) {
parts.push(`[Assistant thinking]: ${thinkingParts.join("\n")}`);
parts.push(`[Think]: ${thinkingParts.join("\n")}`);
}
if (textParts.length > 0) {
parts.push(`[Assistant]: ${textParts.join("\n")}`);
}
if (toolCalls.length > 0) {
parts.push(`[Assistant tool calls]: ${toolCalls.join("; ")}`);
parts.push(`[Tool Call]: ${renderToolCalls(toolCalls, grammar)}`);
}
} else if (msg.role === "toolResult") {
if (uselessCallIds.has(msg.toolCallId)) continue;
@@ -248,7 +246,10 @@ export function serializeConversation(messages: Message[]): string {
.map(c => c.text)
.join("");
if (content) {
parts.push(`[Tool result]: ${truncateForSummary(content, TOOL_RESULT_MAX_CHARS)}`);
const text = truncateForSummary(content, TOOL_RESULT_MAX_CHARS);
parts.push(
`[Tool Result]: ${renderToolResult(msg.toolCallId, msg.toolName, msg.isError === true, text, grammar)}`,
);
}
}
}
@@ -256,6 +257,38 @@ export function serializeConversation(messages: Message[]): string {
return parts.join("\n\n");
}
/**
* Render an assistant turn's tool calls. With a grammar, emit the model's
* native invocation block; otherwise fall back to a compact `name(args)` list.
*/
function renderToolCalls(calls: ToolCall[], grammar: Grammar | undefined): string {
if (grammar) return grammar.renderAssistantToolCalls(calls);
return calls
.map(call => {
const argsStr = Object.entries(call.arguments as Record<string, unknown>)
.map(([k, v]) => `${k}=${JSON.stringify(v)}`)
.join(", ");
return `${call.name}(${argsStr})`;
})
.join("; ");
}
/**
* Render a single tool result. With a grammar, emit the model's native
* tool-result envelope; otherwise return the (already truncated) text verbatim.
*/
function renderToolResult(
id: string,
name: string,
isError: boolean,
text: string,
grammar: Grammar | undefined,
): string {
if (!grammar) return text;
const result: GrammarToolResult = { id, name, index: 0, text, isError };
return grammar.renderToolResults([result]);
}
// ============================================================================
// Summarization System Prompt
// ============================================================================
-1
View File
@@ -6,7 +6,6 @@ export * from "./agent-loop";
export * from "./append-only-context";
// Compaction
export * from "./compaction";
export * from "./harmony-leak";
// Proxy utilities
export * from "./proxy";
// Run-level telemetry collector + aggregators
+32 -1
View File
@@ -17,8 +17,9 @@ import type {
ToolResultMessage,
TSchema,
} from "@oh-my-pi/pi-ai";
import type { ToolCallSyntax } from "@oh-my-pi/pi-ai/grammar";
import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
import type { AppendOnlyContextManager } from "./append-only-context";
import type { HarmonyAuditEvent } from "./harmony-leak";
import type { AgentRunCoverage, AgentRunSummary } from "./run-collector";
import type { AgentTelemetryConfig } from "./telemetry";
@@ -199,6 +200,27 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
* then strips from arguments before executing tools.
*/
intentTracing?: boolean;
/**
* Owned tool calling syntax.
*
* Undefined keeps provider-native tool calling. A syntax value sends no
* native `tools`, forces `toolChoice` off, appends that syntax's tool catalog
* instructions, re-encodes prior tool calls/results as text, and parses the
* model's text output back into canonical `toolCall` blocks.
*/
toolCallSyntax?: ToolCallSyntax;
/**
* When owned (in-band) tool calling is active and the model starts
* fabricating a tool result inside its own turn, control how the loop reacts:
* - `true` (default): abort the provider request immediately so it stops
* generating the hallucinated continuation (cheaper, lower latency).
* - `false`: let the request finish and silently discard everything past the
* fabrication boundary (keeps the connection alive but pays for the tokens
* the model spends on the discarded tail).
* Only meaningful when {@link toolCallSyntax} (or `PI_OWNED_TOOLS`) selects an
* owned syntax; native tool calling never fabricates results in text.
*/
abortOnFabricatedToolResult?: boolean;
/**
* Append-only context mode — stabilizes system prompt + tool spec bytes
* across turns so provider prefix caches hit at maximum rate.
@@ -481,6 +503,15 @@ export interface AgentTool<TParameters extends TSchema = TSchema, TDetails = any
concurrency?: "shared" | "exclusive" | ((args: Partial<Static<TParameters>>) => "shared" | "exclusive");
/** If true, argument validation errors are non-fatal: raw args are passed to execute() instead of returning an error to the LLM. */
lenientArgValidation?: boolean;
/**
* If true, the agent loop may abort this tool mid-execution to deliver a
* queued steering message (instead of waiting for the tool to finish on its
* own). Set only on tools that purely *wait* and observe their abort signal
* cleanly (e.g. the `job` poll), so the abort surfaces the tool's current
* snapshot rather than corrupting a side effect. Honored only when
* `interruptMode` is "immediate".
*/
interruptible?: boolean;
/**
* Controls how the INTENT_FIELD (`_i`) is handled for this tool.
* - `"require"` (default): `_i` is injected and required in the parameter schema.
+143
View File
@@ -842,6 +842,149 @@ describe("agentLoop with AgentMessage", () => {
expect(sawInterruptInContext).toBe(true);
});
it("drains queued steering by aborting an interruptible tool mid-wait", async () => {
const toolSchema = z.object({});
let steerReady = false;
let drained = false;
let observedAbort = false;
let resolvedByTimeout = false;
const tool: AgentTool<typeof toolSchema, Record<string, never>> = {
name: "wait",
label: "Wait",
description: "Blocks until aborted (mimics a job poll)",
parameters: toolSchema,
interruptible: true,
async execute(_toolCallId, _params, signal) {
steerReady = true;
const { promise, resolve } = Promise.withResolvers<void>();
if (signal?.aborted) {
resolve();
} else {
const timer = setTimeout(() => {
resolvedByTimeout = true;
resolve();
}, 2000);
signal?.addEventListener(
"abort",
() => {
clearTimeout(timer);
resolve();
},
{ once: true },
);
}
await promise;
observedAbort = signal?.aborted === true;
return { content: [{ type: "text", text: "waited" }], details: {} };
},
};
const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] };
const mock = createMockModel({
responses: [
{ content: [{ type: "toolCall", id: "tool-1", name: "wait", arguments: {} }] },
{ content: ["done"] },
],
});
const config: AgentLoopConfig = {
model: mock.model,
convertToLlm: identityConverter,
interruptMode: "immediate",
hasSteeringMessages: () => steerReady && !drained,
getSteeringMessages: async () => {
if (steerReady && !drained) {
drained = true;
return [createUserMessage("interrupt")];
}
return [];
},
};
const events: AgentEvent[] = [];
for await (const event of agentLoop([createUserMessage("start")], context, config, undefined, mock.stream)) {
events.push(event);
}
expect(observedAbort).toBe(true);
expect(resolvedByTimeout).toBe(false);
expect(drained).toBe(true);
expect(
events.some(e => e.type === "message_start" && e.message.role === "user" && e.message.content === "interrupt"),
).toBe(true);
});
it("does not abort a non-interruptible tool mid-wait; steering still drains at the boundary", async () => {
const toolSchema = z.object({});
let steerReady = false;
let drained = false;
let observedAbort = false;
let resolvedByTimeout = false;
const tool: AgentTool<typeof toolSchema, Record<string, never>> = {
name: "wait",
label: "Wait",
description: "Blocks on its own window (no interruptible flag)",
parameters: toolSchema,
async execute(_toolCallId, _params, signal) {
steerReady = true;
const { promise, resolve } = Promise.withResolvers<void>();
if (signal?.aborted) {
resolve();
} else {
const timer = setTimeout(() => {
resolvedByTimeout = true;
resolve();
}, 300);
signal?.addEventListener(
"abort",
() => {
clearTimeout(timer);
resolve();
},
{ once: true },
);
}
await promise;
observedAbort = signal?.aborted === true;
return { content: [{ type: "text", text: "waited" }], details: {} };
},
};
const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] };
const mock = createMockModel({
responses: [
{ content: [{ type: "toolCall", id: "tool-1", name: "wait", arguments: {} }] },
{ content: ["done"] },
],
});
const config: AgentLoopConfig = {
model: mock.model,
convertToLlm: identityConverter,
interruptMode: "immediate",
hasSteeringMessages: () => steerReady && !drained,
getSteeringMessages: async () => {
if (steerReady && !drained) {
drained = true;
return [createUserMessage("interrupt")];
}
return [];
},
};
const events: AgentEvent[] = [];
for await (const event of agentLoop([createUserMessage("start")], context, config, undefined, mock.stream)) {
events.push(event);
}
expect(observedAbort).toBe(false);
expect(resolvedByTimeout).toBe(true);
expect(drained).toBe(true);
expect(
events.some(e => e.type === "message_start" && e.message.role === "user" && e.message.content === "interrupt"),
).toBe(true);
});
it("leaves steering queued when the run is aborted while interrupted tools settle", async () => {
// Regression: the mid-batch steering poll used to DEQUEUE the message into
// a loop-local variable. An external abort while the in-flight tools were
@@ -1,7 +1,7 @@
import { describe, expect, it } from "bun:test";
import { AppendOnlyContextManager, AppendOnlyLog, StablePrefix } from "@oh-my-pi/pi-agent-core/append-only-context";
import type { AgentContext, AgentTool } from "@oh-my-pi/pi-agent-core/types";
import type { Message, Tool } from "@oh-my-pi/pi-ai";
import type { Message, Tool, ToolExample } from "@oh-my-pi/pi-ai";
// ---------------------------------------------------------------------------
// Helpers
@@ -16,12 +16,18 @@ function makeContext(overrides?: Partial<AgentContext>): AgentContext {
};
}
function makeTool(name: string, description?: string, parameters?: Record<string, unknown>): AgentTool {
function makeTool(
name: string,
description?: string,
parameters?: Record<string, unknown>,
examples?: readonly ToolExample[],
): AgentTool {
return {
name,
description: description ?? `Tool ${name}`,
parameters: parameters ?? { type: "object", properties: {} },
label: name,
examples,
execute: async () => ({ content: [{ type: "text", text: "done" }] }),
} as AgentTool;
}
@@ -629,6 +635,61 @@ describe("intent injection through build()", () => {
});
});
describe("tool examples injection through build()", () => {
const findExamples: readonly ToolExample[] = [{ caption: "Find files", call: { paths: ["src/**/*.ts"] } }];
const findParams = {
type: "object",
properties: { paths: { type: "array", items: { type: "string" } } },
};
it("injects examples when exampleSyntax is provided", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("find", "Find files.", findParams, findExamples);
const ctx = makeContext({ tools: [tool] });
const result = mgr.build(ctx, { intentTracing: false, exampleSyntax: "anthropic" });
const desc = result.tools?.[0]?.description ?? "";
expect(desc).toContain("<examples>");
expect(desc).toContain("# Find files");
expect(desc).toContain('<invoke name="find">');
});
it("omits examples when exampleSyntax is undefined", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("find", "Find files.", findParams, findExamples);
const ctx = makeContext({ tools: [tool] });
const result = mgr.build(ctx, { intentTracing: false });
const desc = result.tools?.[0]?.description ?? "";
expect(desc).toBe("Find files.");
});
it("injects the `_i` placeholder into examples when intentTracing is on", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("find", "Find files.", findParams, findExamples);
const ctx = makeContext({ tools: [tool] });
const result = mgr.build(ctx, { intentTracing: true, exampleSyntax: "anthropic" });
const desc = result.tools?.[0]?.description ?? "";
expect(desc).toContain('<parameter name="_i"');
expect(desc).toContain("…");
});
it("exampleSyntax flip invalidates the fingerprint cache", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("find", "Find files.", undefined, findExamples);
const ctx = makeContext({ tools: [tool] });
mgr.build(ctx, { intentTracing: false });
const fpNoExamples = mgr.prefix.fingerprint;
mgr.build(ctx, { intentTracing: false, exampleSyntax: "anthropic" });
const fpWithExamples = mgr.prefix.fingerprint;
expect(fpNoExamples).not.toBe(fpWithExamples);
});
});
// ---------------------------------------------------------------------------
// Tool-call mutation detection
// ---------------------------------------------------------------------------
@@ -0,0 +1,148 @@
import { describe, expect, it } from "bun:test";
import { agentLoop } from "@oh-my-pi/pi-agent-core/agent-loop";
import type { AgentContext, AgentLoopConfig, AgentMessage, AgentTool } from "@oh-my-pi/pi-agent-core/types";
import type { AssistantMessage, Context, Message, TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { z } from "zod/v4";
import { createUserMessage } from "./helpers";
function identityConverter(messages: AgentMessage[]): Message[] {
return messages.filter(m => m.role === "user" || m.role === "assistant" || m.role === "toolResult") as Message[];
}
function wireText(message: Message): string {
if (typeof message.content === "string") return message.content;
return (message.content as (TextContent | { type: string })[])
.map(b => (b.type === "text" ? (b as TextContent).text : ""))
.join("");
}
describe("agentLoop with owned in-band tool calls", () => {
it("executes <tool_call> text, strips native tools from the wire, and re-encodes history as text", async () => {
const echoArgs: Array<{ msg: string }> = [];
const toolSchema = z.object({ msg: z.string().describe("message to echo") });
const echoTool: AgentTool<typeof toolSchema, { msg: string }> = {
name: "echo",
label: "Echo",
description: "Echo a message back",
parameters: toolSchema,
async execute(_toolCallId, params) {
echoArgs.push(params);
return { content: [{ type: "text", text: `echoed:${params.msg}` }], details: params };
},
};
const captured: Context[] = [];
const mock = createMockModel({
responses: [
context => {
captured.push(context);
return {
content: [
"on it\n<tool_call>echo\n<arg_key>msg</arg_key>\n<arg_value>hello world</arg_value>\n</tool_call>",
],
};
},
context => {
captured.push(context);
return { content: ["all done"] };
},
],
});
const context: AgentContext = { systemPrompt: ["BASE PROMPT"], messages: [], tools: [echoTool] };
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter, toolCallSyntax: "glm" };
const messages = await agentLoop([createUserMessage("say hi")], context, config, undefined, mock.stream).result();
// The tool was actually executed with the parsed (verbatim) argument.
expect(echoArgs).toEqual([{ msg: "hello world" }]);
expect(captured).toHaveLength(2);
// First request: no native tools on the wire; catalog + grammar injected.
expect(captured[0].tools).toBeUndefined();
const sys0 = captured[0].systemPrompt ?? [];
expect(sys0[0]).toBe("BASE PROMPT");
const promptSection = sys0.join("\n");
expect(promptSection).toContain("<tools>");
expect(promptSection).toContain('"name":"echo"');
expect(promptSection).toContain("YOU MUST EMIT THE STOP SEQUENCE AND HALT");
// Second request: the wire carries NO native tool blocks — prior call/result
// are plain <tool_call> / <tool_response> text, and tools are still stripped.
const wire2 = captured[1].messages;
expect(captured[1].tools).toBeUndefined();
for (const m of wire2) {
expect(m.role).not.toBe("toolResult");
if (m.role === "assistant") {
expect((m.content as { type: string }[]).some(b => b.type === "toolCall")).toBe(false);
}
}
const wireAssistant = wire2.find(m => m.role === "assistant");
expect(wireAssistant).toBeDefined();
const at = wireText(wireAssistant!);
expect(at).toContain("on it");
expect(at).toContain("<tool_call>echo");
expect(at).toContain("<arg_value>hello world</arg_value>");
const resultsText = wire2
.filter(m => m.role === "user")
.map(wireText)
.join("\n");
expect(resultsText).toContain("<tool_response>");
expect(resultsText).toContain("echoed:hello world");
// The internal store stays canonical: native toolCall block + toolResult message.
const internalAssistant = messages.find(
(m): m is AssistantMessage => m.role === "assistant" && m.content.some(b => b.type === "toolCall"),
);
expect(internalAssistant).toBeDefined();
const internalResult = messages.find((m): m is ToolResultMessage => m.role === "toolResult");
expect(internalResult).toBeDefined();
expect(internalResult!.toolName).toBe("echo");
expect(wireText(internalResult!)).toBe("echoed:hello world");
});
it("executes Hermes/Qwen JSON tool calls when that syntax is selected", async () => {
const echoArgs: Array<{ msg: string }> = [];
const toolSchema = z.object({ msg: z.string().describe("message to echo") });
const echoTool: AgentTool<typeof toolSchema, { msg: string }> = {
name: "echo",
label: "Echo",
description: "Echo a message back",
parameters: toolSchema,
async execute(_toolCallId, params) {
echoArgs.push(params);
return { content: [{ type: "text", text: `echoed:${params.msg}` }], details: params };
},
};
const captured: Context[] = [];
const mock = createMockModel({
responses: [
context => {
captured.push(context);
return { content: ['<tool_call>\n{"name":"echo","arguments":{"msg":"hi"}}\n</tool_call>'] };
},
context => {
captured.push(context);
return { content: ["done"] };
},
],
});
const context: AgentContext = { systemPrompt: ["BASE PROMPT"], messages: [], tools: [echoTool] };
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter, toolCallSyntax: "hermes" };
await agentLoop([createUserMessage("say hi")], context, config, undefined, mock.stream).result();
expect(echoArgs).toEqual([{ msg: "hi" }]);
expect(captured[0].tools).toBeUndefined();
expect((captured[0].systemPrompt ?? []).join("\n")).toContain('"name":"function_name","arguments"');
const resultsText = captured[1].messages
.filter(m => m.role === "user")
.map(wireText)
.join("\n");
expect(resultsText).toContain("<tool_response>");
expect(resultsText).toContain("echoed:hi");
});
});
@@ -60,6 +60,6 @@ describe("serializeConversation — useless pairs", () => {
]);
expect(out).toContain('pattern="beta"');
expect(out).toContain("[Tool result]: grep crashed");
expect(out).toContain("[Tool Result]: grep crashed");
});
});
@@ -503,3 +503,22 @@ describe("pruneToolOutputs — useless results", () => {
expect(resultText(result1)).toBe(NO_MATCH_TEXT);
});
});
describe("pruneToolOutputs — small-result floor", () => {
test("sub-floor results are left intact while a large neighbor is pruned", () => {
// "ok" is ~1 token: blanking it to `[Output truncated - 1 tokens]` would
// grow the context, so the floor must keep it. The large neighbor still prunes.
const [tinyCall, tinyResult] = readPair("src/tiny.ts", "ok", T0);
const [bigCall, bigResult] = readPair("src/big.ts", FILE_CONTENT, T0 + 1_000);
const entries: SessionEntry[] = [tinyCall, tinyResult, bigCall, bigResult];
// Protect window empty and zero savings threshold: only size keeps the tiny one.
const result = pruneToolOutputs(entries, { protectTokens: 0, minimumSavings: 0, protectedTools: [] });
expect(result.prunedCount).toBe(1);
expect(resultText(tinyResult)).toBe("ok");
expect(resultMessage(tinyResult).prunedAt).toBeUndefined();
expect(resultText(bigResult)).toMatch(/^\[Output truncated - \d+ tokens\]$/);
expect(resultMessage(bigResult).prunedAt).toBeDefined();
});
});
+1 -1
View File
@@ -53,7 +53,7 @@ function readResult(toolCallId: string, text: string): SessionMessageEntry {
describe("conditional tool-result protection", () => {
it("prunes regular read results but keeps skill:// reads", () => {
const skillResult = readResult("skill-read", "skill read output that must remain intact");
const fileResult = readResult("file-read", "file read output that can be pruned");
const fileResult = readResult("file-read", "file read output that can be pruned ".repeat(20));
const entries = [
assistantReadCall("skill-read", "skill://session-memory"),
skillResult,
+52
View File
@@ -5,6 +5,58 @@
### Fixed
- Fixed OpenAI Responses, Azure OpenAI Responses, and Codex Responses providers ignoring async `onPayload` replacement bodies. Provider payload hooks can now transform the actual request body sent upstream, matching the Anthropic/Gemini replacement contract.
## [15.13.3] - 2026-06-15
### Added
- Added the `gemini` in-band tool-call syntax with Python-style ```tool_code``` blocks and `default_api` invocations
- Added the `gemma` token-delimited in-band tool-call syntax using `<|tool_call>` and `<|tool_response>` blocks
- Added `gemini` and `gemma` to owned stream tool-result token detection so their tool responses are recognized
- Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming
- Added the Azure OpenAI provider definition (`azure`) to the registry; `AZURE_OPENAI_API_KEY` resolves as its env-var API key via the catalog provider table.
### Changed
- Gemini tool-call examples now render without the `default_api.` namespace prefix, keeping `<example>` blocks concise. The live wire format still uses `default_api.` per the Gemini grammar.
### Fixed
- Fixed duplicate tool call projections by deduplicating provider-native `toolCall` events against in-band `tool_code` calls and keeping only the first real channel
- Dropped nameless native `toolCall` events so they no longer appear as surfaced tool calls in owned-mode streams
- Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming
- Fixed Gemini/Gemma in-band tool-call parsing around Python comments, raw/unicode string literals, and Gemma close-token text inside string values.
## [15.13.2] - 2026-06-15
### Added
- Added `jsonSchemaToTypeScript` to `@oh-my-pi/pi-ai/utils/schema` to render JSON Schema argument shapes as compact, human-readable TypeScript-style signatures
- Added the generic `ToolExample` type (`ToolCallExample`/`ToolCompareExample`/`ToolNoteExample`, parameterized over a tool's argument shape) and an `examples` property on the `Tool` interface for defining tool-call examples once as data.
- Added `renderToolExamples` (via `@oh-my-pi/pi-ai/grammar`) to render a tool's examples into an `<examples>` block in the model's native tool-call syntax, with an optional `_i` intent-field placeholder injected when intent tracing is active.
- Added per-grammar `renderToolCall` rendering of a single tool-call invocation (the inner element only, without the parallel-call block envelope), distinct from `renderAssistantToolCalls` which renders a complete block of one or more parallel calls.
- Added a `GrammarRenderOptions.example` flag to `renderToolCall`: when set, the invocation renders as the bare payload — Harmony emits just the JSON arguments, dropping the verbose `<|start|>…<|message|>…<|call|>` envelope — so `renderToolExamples` keeps `<examples>` blocks legible.
- Added an `abortOnFabrication` parameter to `wrapInbandToolStream` (default `true`): when `false`, a fabricated in-band tool-result continuation is discarded without aborting the provider request instead of cutting the turn short.
- Added `@oh-my-pi/pi-ai/utils/harmony-leak` export with helpers to detect, audit, and recover GPT-5 Harmony tool-call header leaks
- Added the `@oh-my-pi/pi-ai/grammar` public entrypoint for grammar factories, prompt/call rendering, in-band scanning, history encoding, and related typed utilities
- Added a unified in-band tool-call grammar engine with syntax-owned scanners, prompts, history rendering, tool-result rendering, and stream adaptation for GLM, Hermes/Qwen, Kimi, XML/Anthropic, DeepSeek, Harmony, and pi-native formats.
### Changed
- Changed Harmony in-band tool-call rendering to omit the `<|constrain|>json` marker before the payload in `commentary` channel calls
- Changed tool inventory rendering to present each tool’s `Parameters` section as a simplified TypeScript-style signature derived from its wire schema
- Added raw in-band tool-call block capture to parsed owned tool calls so debugging can inspect the exact model-emitted call syntax.
- Moved the canonical `ToolCallSyntax` union to `@oh-my-pi/pi-catalog/identity` and re-exported it from `@oh-my-pi/pi-ai/grammar` so the catalog can own the syntax vocabulary without an `@oh-my-pi/pi-ai` runtime import; all existing import paths are unchanged.
- Made tool-call argument validation more lenient for schema-directed scalar coercions, including object/array stringification and 0/1 boolean coercion.
- Changed `renderToolInventory` (the verbose system-prompt inventory and `/dump`) to render each tool as a `# Tool: <name>` markdown section instead of a `<tool name="…">…</tool>` wrapper.
### Fixed
- Fixed Harmony leak handling support by adding `recoverHarmonyToolCall` plus leak-detection workflows for contaminated assistant messages so recoverable tool-call arguments can be safely truncated and retried
- Fixed false-positive gating in Harmony leak heuristics using signal-based checks so unrelated text containing `to=functions...` is not treated as leaked tool-call markup
- Routed Kimi, DeepSeek DSML, and plain thinking markup healing through the shared in-band scanners so provider leak repair and owned tool calling parse the same wire formats.
- Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the *cumulative* args text so far per `agent.proto`, but the handler concatenated each snapshot onto the buffer, garbling the JSON; (2) `tool_call_completed` carries an `McpArgs` map that omits oversized parameters entirely and downgrades unparsable values to their raw string fallback, but the handler unconditionally overwrote the streamed args with that map. The handler now strips the already-buffered prefix from each `args_text_delta` snapshot (falling back to append when the snapshot doesn't extend the buffer) and merges the decoded `McpArgs` map into the streamed args — preserving streamed keys the completion frame omits and the structured value when the completion frame downgrades to a string. ([#2615](https://github.com/can1357/oh-my-pi/issues/2615))
- Fixed Codex Responses stream mis-routing interleaved `function_call_arguments.delta` events when more than one tool call was open concurrently. The runtime tracked a singleton `currentItem`/`currentBlock`, so every delta — regardless of `item_id` — was appended to whichever item was most recently added, and `output_item.done` for the earlier call then overwrote a sibling's stored arguments (visible as `tasks: Invalid input: expected array, received undefined` on the `task` tool). Open items are now keyed by `item_id` with `output_index` fallback; deltas/done events route to the matching block, late deltas whose item already closed are dropped instead of corrupting a sibling, and `toolcall_*` stream events emit the right `contentIndex` per call ([#2619](https://github.com/can1357/oh-my-pi/issues/2619)).
## [15.13.1] - 2026-06-15
### Fixed
+9 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "15.13.1",
"version": "15.13.3",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -91,6 +91,14 @@
"types": "./src/usage/*.ts",
"import": "./src/usage/*.ts"
},
"./utils/harmony-leak": {
"types": "./src/utils/harmony-leak.ts",
"import": "./src/utils/harmony-leak.ts"
},
"./grammar": {
"types": "./src/grammar/index.ts",
"import": "./src/grammar/index.ts"
},
"./utils/*": {
"types": "./src/utils/*.ts",
"import": "./src/utils/*.ts"
+31
View File
@@ -0,0 +1,31 @@
## Format guide
A call is a `<function_calls>` block wrapping one or more `<invoke>` blocks, each holding `<parameter>` children:
```text
<function_calls>
<invoke name="tool_name"><parameter name="arg_name">arg value</parameter></invoke>
</function_calls>
```
Results arrive later in a `<function_results>` block, one `<result>` per call (failures use `<error>` with `<stderr>` in place of `<result>` with `<stdout>`):
```text
<function_results>
<result>
<tool_name>tool_name</tool_name>
<stdout>verbatim tool result</stdout>
</result>
</function_results>
```
## Rules
- `name` MUST match a listed function.
- String/scalar parameters: exact text, spaces preserved. Lists/objects: JSON.
- Multiple calls: multiple `<invoke>` blocks in one `<function_calls>`.
- You MAY write visible text before the calls.
- NEVER emit `tool_calls` JSON.
- NEVER use the legacy `<tool_name>`/`<parameters>` call syntax.
- Read each `<result>`/`<error>` in call order. NEVER emit `<function_results>` yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+521
View File
@@ -0,0 +1,521 @@
import { parseJsonWithRepair } from "../utils/json-parse";
import grammarPrompt from "./anthropic.md" with { type: "text" };
import { buildStringArgsResolver, mintToolCallId } from "./coercion";
import { renderAnthropicInvocation, renderAnthropicToolCalls, renderAnthropicToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
const MAX_PARTIAL_TAG_LENGTH = 256;
const MAX_PARAMETER_VALUE_LENGTH = 1_000_000;
const WRAPPER_TAGS: Record<string, true> = { function_calls: true, tool_calls: true };
const THINKING_TAGS: Record<string, true> = { thinking: true, think: true, scratchpad: true };
const BASE_TAG_PREFIXES = [
"<function_calls",
"</function_calls",
"<tool_calls",
"</tool_calls",
"<invoke",
"</invoke",
"<parameter",
"</parameter",
"<antml:function_calls",
"</antml:function_calls",
"<antml:tool_calls",
"</antml:tool_calls",
"<antml:invoke",
"</antml:invoke",
"<antml:parameter",
"</antml:parameter",
] as const;
const THINKING_TAG_PREFIXES = [
"<thinking",
"</thinking",
"<think",
"</think",
"<scratchpad",
"</scratchpad",
"<antml:thinking",
"</antml:thinking",
"<antml:think",
"</antml:think",
"<antml:scratchpad",
"</antml:scratchpad",
] as const;
type ScannerState = "outside" | "section" | "invoke" | "parameter" | "thinking";
type ReturnState = "outside" | "section";
interface ParsedTag {
readonly raw: string;
readonly localName: string;
readonly prefix: string;
readonly closing: boolean;
readonly selfClosing: boolean;
readonly attrs: ReadonlyMap<string, string>;
}
type TagRead = ParsedTag | "partial" | undefined;
export class AnthropicInbandScanner implements InbandScanner {
#buffer = "";
#state: ScannerState = "outside";
#returnState: ReturnState = "outside";
#afterThinkingState: ReturnState = "outside";
#id = "";
#name = "";
#args: Record<string, unknown> = {};
#started = false;
#paramName = "";
#paramValue = "";
#paramString: boolean | undefined;
#paramTruncated = false;
#paramClosePrefixes: readonly string[] = [];
#rawBlock = "";
#thinking = "";
#thinkingTag = "";
#thinkingClosePrefixes: readonly string[] = [];
readonly #stringArgs: (toolName: string) => ReadonlySet<string>;
readonly #parseThinking: boolean;
constructor(options: InbandScannerOptions = {}) {
this.#stringArgs = options.stringArgs ?? buildStringArgsResolver(options.tools);
this.#parseThinking = options.parseThinking === true;
}
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
let progressed: boolean;
switch (this.#state) {
case "outside":
progressed = this.#consumeOutside(final, events);
break;
case "section":
progressed = this.#consumeSection(final, events);
break;
case "invoke":
progressed = this.#consumeInvoke(final, events);
break;
case "parameter":
progressed = this.#consumeParameter(final);
break;
case "thinking":
progressed = this.#consumeThinking(final, events);
break;
}
if (!progressed) break;
}
if (final) this.#flushFinal(events);
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): boolean {
const tagStart = this.#buffer.indexOf("<");
if (tagStart === -1) {
this.#emitText(this.#buffer, events);
this.#buffer = "";
return false;
}
if (tagStart > 0) {
this.#emitText(this.#buffer.slice(0, tagStart), events);
this.#buffer = this.#buffer.slice(tagStart);
return true;
}
const tag = this.#peekTag(final, this.#relevantPrefixes());
if (tag === "partial") return false;
if (!tag) {
this.#emitText(this.#buffer[0]!, events);
this.#buffer = this.#buffer.slice(1);
return true;
}
if (!tag.closing && WRAPPER_TAGS[tag.localName] === true) {
this.#buffer = this.#buffer.slice(tag.raw.length);
this.#state = "section";
return true;
}
if (!tag.closing && tag.localName === "invoke") {
this.#buffer = this.#buffer.slice(tag.raw.length);
this.#startInvoke(tag, "outside", events);
return true;
}
if (this.#isThinkingOpen(tag)) {
this.#buffer = this.#buffer.slice(tag.raw.length);
this.#startThinking(tag, "outside", events);
return true;
}
if (tag.closing && WRAPPER_TAGS[tag.localName] === true) {
this.#buffer = this.#buffer.slice(tag.raw.length);
return true;
}
this.#emitText(this.#buffer[0]!, events);
this.#buffer = this.#buffer.slice(1);
return true;
}
#consumeSection(final: boolean, events: InbandScanEvent[]): boolean {
const tagStart = this.#buffer.indexOf("<");
if (tagStart === -1) {
this.#buffer = "";
return false;
}
if (tagStart > 0) {
this.#buffer = this.#buffer.slice(tagStart);
return true;
}
const tag = this.#peekTag(final, this.#relevantPrefixes());
if (tag === "partial") return false;
if (!tag) {
this.#buffer = this.#buffer.slice(1);
return true;
}
this.#buffer = this.#buffer.slice(tag.raw.length);
if (tag.closing && WRAPPER_TAGS[tag.localName] === true) {
this.#state = "outside";
return true;
}
if (!tag.closing && tag.localName === "invoke") {
this.#startInvoke(tag, "section", events);
return true;
}
if (this.#parseThinking && !tag.closing && THINKING_TAGS[tag.localName] === true) {
this.#startThinking(tag, "section", events);
}
return true;
}
#consumeInvoke(final: boolean, events: InbandScanEvent[]): boolean {
const tagStart = this.#buffer.indexOf("<");
if (tagStart === -1) {
if (final) this.#resetCall(this.#returnState);
else {
this.#rawBlock += this.#buffer;
this.#buffer = "";
}
return false;
}
if (tagStart > 0) {
const consumed = this.#buffer.slice(0, tagStart);
this.#rawBlock += consumed;
this.#buffer = this.#buffer.slice(tagStart);
return true;
}
const tag = this.#peekTag(final, this.#relevantPrefixes());
if (tag === "partial") return false;
if (!tag) {
const consumed = this.#buffer[0]!;
this.#rawBlock += consumed;
this.#buffer = this.#buffer.slice(1);
return true;
}
this.#rawBlock += tag.raw;
this.#buffer = this.#buffer.slice(tag.raw.length);
if (tag.closing && tag.localName === "invoke") {
if (this.#started) {
events.push({
type: "toolEnd",
id: this.#id,
name: this.#name,
arguments: this.#args,
rawBlock: this.#rawBlock,
});
}
this.#resetCall(this.#returnState);
return true;
}
if (!tag.closing && tag.localName === "parameter") {
this.#startParameter(tag);
if (tag.selfClosing) this.#finishParameter();
return true;
}
return true;
}
#consumeParameter(final: boolean): boolean {
const tagStart = this.#buffer.indexOf("<");
if (tagStart === -1) {
if (final) {
this.#resetCall(this.#returnState);
this.#buffer = "";
return false;
}
this.#appendParameterValue(this.#buffer);
this.#rawBlock += this.#buffer;
this.#buffer = "";
return false;
}
if (tagStart > 0) {
const consumed = this.#buffer.slice(0, tagStart);
this.#appendParameterValue(consumed);
this.#rawBlock += consumed;
this.#buffer = this.#buffer.slice(tagStart);
return true;
}
const tag = this.#peekTag(final, this.#paramClosePrefixes);
if (tag === "partial") return false;
if (tag?.closing && tag.localName === "parameter") {
this.#rawBlock += tag.raw;
this.#buffer = this.#buffer.slice(tag.raw.length);
this.#finishParameter();
return true;
}
if (final && !tag) {
this.#resetCall(this.#returnState);
this.#buffer = "";
return false;
}
const consumed = this.#buffer[0]!;
this.#appendParameterValue(consumed);
this.#rawBlock += consumed;
this.#buffer = this.#buffer.slice(1);
return true;
}
#consumeThinking(final: boolean, events: InbandScanEvent[]): boolean {
const tagStart = this.#buffer.indexOf("<");
if (tagStart === -1) {
if (final) {
this.#appendThinking(this.#buffer, events);
this.#buffer = "";
this.#finishThinking(events);
return false;
}
this.#appendThinking(this.#buffer, events);
this.#buffer = "";
return false;
}
if (tagStart > 0) {
this.#appendThinking(this.#buffer.slice(0, tagStart), events);
this.#buffer = this.#buffer.slice(tagStart);
return true;
}
const tag = this.#peekTag(final, this.#thinkingClosePrefixes);
if (tag === "partial") return false;
if (tag?.closing && tag.localName === this.#thinkingTag) {
this.#buffer = this.#buffer.slice(tag.raw.length);
this.#finishThinking(events);
return true;
}
if (final && !tag) {
this.#appendThinking(this.#buffer, events);
this.#buffer = "";
this.#finishThinking(events);
return false;
}
this.#appendThinking(this.#buffer[0]!, events);
this.#buffer = this.#buffer.slice(1);
return true;
}
#flushFinal(events: InbandScanEvent[]): void {
if (this.#state === "outside") return;
if (this.#state === "thinking") this.#finishThinking(events);
else this.#resetCall(this.#returnState);
this.#state = "outside";
this.#buffer = "";
}
#startInvoke(tag: ParsedTag, returnState: ReturnState, events: InbandScanEvent[]): void {
this.#returnState = returnState;
this.#id = mintToolCallId();
this.#name = tag.attrs.get("name")?.trim() ?? "";
this.#args = {};
this.#rawBlock = tag.raw;
this.#started = this.#name.length > 0;
this.#state = "invoke";
if (this.#started) events.push({ type: "toolStart", id: this.#id, name: this.#name });
}
#startParameter(tag: ParsedTag): void {
this.#paramName = tag.attrs.get("name")?.trim() ?? "";
this.#paramValue = "";
this.#paramTruncated = false;
this.#paramString = parseStringAttribute(tag.attrs.get("string"));
this.#paramClosePrefixes = closePrefixes("parameter", tag.prefix);
this.#state = "parameter";
}
#appendParameterValue(delta: string): void {
if (delta.length === 0) return;
const remaining = MAX_PARAMETER_VALUE_LENGTH - this.#paramValue.length;
if (remaining > 0) this.#paramValue += delta.slice(0, remaining);
if (delta.length > remaining) this.#paramTruncated = true;
}
#finishParameter(): void {
if (this.#paramName.length > 0) {
const value = this.#paramTruncated
? `${this.#paramValue}\n…[parameter truncated: exceeded ${MAX_PARAMETER_VALUE_LENGTH} bytes]`
: this.#paramValue;
this.#args[this.#paramName] = this.#coerceParameterValue(this.#paramName, value, this.#paramString);
}
this.#paramName = "";
this.#paramValue = "";
this.#paramString = undefined;
this.#paramTruncated = false;
this.#paramClosePrefixes = [];
this.#state = "invoke";
}
#coerceParameterValue(name: string, raw: string, explicitString: boolean | undefined): unknown {
if (explicitString ?? this.#stringArgs(this.#name).has(name)) return raw;
const trimmed = raw.trim();
if (trimmed.length === 0) return raw;
try {
return parseJsonWithRepair<unknown>(trimmed);
} catch {
return raw;
}
}
#startThinking(tag: ParsedTag, afterState: ReturnState, events: InbandScanEvent[]): void {
this.#afterThinkingState = afterState;
this.#thinking = "";
this.#thinkingTag = tag.localName;
this.#thinkingClosePrefixes = closePrefixes(tag.localName, tag.prefix);
this.#state = "thinking";
events.push({ type: "thinkingStart" });
if (tag.selfClosing) this.#finishThinking(events);
}
#appendThinking(delta: string, events: InbandScanEvent[]): void {
if (delta.length === 0) return;
this.#thinking += delta;
events.push({ type: "thinkingDelta", delta });
}
#finishThinking(events: InbandScanEvent[]): void {
events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#thinkingTag = "";
this.#thinkingClosePrefixes = [];
this.#state = this.#afterThinkingState;
this.#afterThinkingState = "outside";
}
#resetCall(nextState: ReturnState): void {
this.#id = "";
this.#name = "";
this.#args = {};
this.#started = false;
this.#paramName = "";
this.#paramValue = "";
this.#paramString = undefined;
this.#paramTruncated = false;
this.#paramClosePrefixes = [];
this.#rawBlock = "";
this.#state = nextState;
}
#peekTag(final: boolean, relevantPrefixes: readonly string[]): TagRead {
const close = this.#buffer.indexOf(">");
if (close === -1) {
if (
!final &&
this.#buffer.length <= MAX_PARTIAL_TAG_LENGTH &&
couldBeTagPrefix(this.#buffer, relevantPrefixes)
) {
return "partial";
}
return undefined;
}
const raw = this.#buffer.slice(0, close + 1);
return parseTag(raw);
}
#isThinkingOpen(tag: ParsedTag): boolean {
if (!this.#parseThinking || tag.closing) return false;
return THINKING_TAGS[tag.localName] === true;
}
#relevantPrefixes(): readonly string[] {
return this.#parseThinking ? ALL_TAG_PREFIXES : BASE_TAG_PREFIXES;
}
#emitText(text: string, events: InbandScanEvent[]): void {
if (text.length > 0) events.push({ type: "text", text });
}
}
const ALL_TAG_PREFIXES = [...BASE_TAG_PREFIXES, ...THINKING_TAG_PREFIXES] as const;
function parseTag(raw: string): ParsedTag | undefined {
const match = /^<\s*(\/?)\s*(?:(?<prefix>[A-Za-z_][\w.-]*):)?(?<localName>[A-Za-z_][\w.-]*)(?<attrs>[^>]*)>$/s.exec(
raw,
);
const localName = match?.groups?.localName;
if (!match || !localName) return undefined;
const attrsText = match.groups?.attrs ?? "";
return {
raw,
localName: localName.toLowerCase(),
prefix: match.groups?.prefix ?? "",
closing: match[1] === "/",
selfClosing: match[1] !== "/" && /\/\s*$/.test(attrsText),
attrs: parseAttributes(attrsText),
};
}
function parseAttributes(text: string): ReadonlyMap<string, string> {
const attrs = new Map<string, string>();
const pattern = /([A-Za-z_:][\w:.-]*)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'<>/=]+))/g;
for (const match of text.matchAll(pattern)) {
const rawName = match[1];
if (!rawName) continue;
const colon = rawName.lastIndexOf(":");
const name = (colon === -1 ? rawName : rawName.slice(colon + 1)).toLowerCase();
attrs.set(name, match[2] ?? match[3] ?? match[4] ?? "");
}
return attrs;
}
function parseStringAttribute(value: string | undefined): boolean | undefined {
if (value === undefined) return undefined;
const normalized = value.trim().toLowerCase();
if (normalized === "false" || normalized === "0" || normalized === "no") return false;
return true;
}
function closePrefixes(localName: string, prefix: string): readonly string[] {
const unprefixed = `</${localName}`;
const antml = `</antml:${localName}`;
if (prefix.length === 0 || prefix === "antml") return [unprefixed, antml];
return [`</${prefix}:${localName}`, unprefixed, antml];
}
function couldBeTagPrefix(buffer: string, prefixes: readonly string[]): boolean {
if (!buffer.startsWith("<")) return false;
for (const prefix of prefixes) {
if (prefix.startsWith(buffer) || buffer.startsWith(prefix)) return true;
}
return false;
}
const grammar: Grammar = {
syntax: "anthropic",
prompt: grammarPrompt,
createScanner: options => new AnthropicInbandScanner(options),
renderToolCall: renderAnthropicInvocation,
renderAssistantToolCalls: renderAnthropicToolCalls,
renderToolResults: renderAnthropicToolResults,
};
export default grammar;
+27
View File
@@ -0,0 +1,27 @@
import { toolWireSchema } from "../utils/schema";
import { getInbandGrammar } from "./factory";
import promptTemplate from "./prompt-template.md" with { type: "text" };
import type { InbandTool, ToolCallSyntax } from "./types";
const TOOLS_TOKEN = "{{TOOLS}}";
const GRAMMAR_TOKEN = "{{GRAMMAR}}";
export function renderToolCatalog(tools: readonly InbandTool[]): string {
return tools
.map(tool =>
JSON.stringify({
type: "function",
function: {
name: tool.name,
description: tool.description ?? "",
parameters: toolWireSchema(tool),
},
}),
)
.join("\n");
}
export function renderInbandToolPrompt(tools: readonly InbandTool[], syntax: ToolCallSyntax): string {
const prompt = getInbandGrammar(syntax).prompt.trim();
return promptTemplate.replace(TOOLS_TOKEN, () => renderToolCatalog(tools)).replace(GRAMMAR_TOKEN, () => prompt);
}
+136
View File
@@ -0,0 +1,136 @@
import { toolWireSchema } from "../utils/schema";
import type { InbandTool } from "./types";
export interface ToolArgShape {
stringArgs: Set<string>;
properties: Record<string, unknown>;
parameterOrder: string[];
}
export function buildArgShapes(tools: readonly InbandTool[] = []): Map<string, ToolArgShape> {
const shapes = new Map<string, ToolArgShape>();
for (const tool of tools) {
const schema = resolveToolSchema(tool);
const props = schema.properties;
const properties =
props && typeof props === "object" && !Array.isArray(props) ? (props as Record<string, unknown>) : {};
const stringArgs = new Set<string>();
const parameterOrder: string[] = [];
for (const key in properties) {
parameterOrder.push(key);
if (isStringOnlySchema(properties[key])) stringArgs.add(key);
}
shapes.set(tool.name, { stringArgs, properties, parameterOrder });
}
return shapes;
}
export function buildStringArgsResolver(tools: readonly InbandTool[] = []): (toolName: string) => ReadonlySet<string> {
const shapes = buildArgShapes(tools);
const empty = new Set<string>();
return (toolName: string) => shapes.get(toolName)?.stringArgs ?? empty;
}
export function resolveToolSchema(tool: InbandTool): Record<string, unknown> {
try {
return toolWireSchema(tool);
} catch {
const params = tool.parameters;
return params && typeof params === "object" && !Array.isArray(params) ? (params as Record<string, unknown>) : {};
}
}
export function isStringOnlySchema(schema: unknown): boolean {
const types = collectSchemaTypes(schema);
types.delete("null");
return types.size === 1 && types.has("string");
}
export function collectSchemaTypes(schema: unknown, out: Set<string> = new Set(), depth = 0): Set<string> {
if (depth > 8 || !schema || typeof schema !== "object" || Array.isArray(schema)) return out;
const node = schema as Record<string, unknown>;
const type = node.type;
if (typeof type === "string") out.add(type);
else if (Array.isArray(type)) for (const t of type) if (typeof t === "string") out.add(t);
if (type === undefined && Array.isArray(node.enum)) {
for (const value of node.enum) out.add(jsonTypeOf(value));
}
if (type === undefined && "const" in node) out.add(jsonTypeOf(node.const));
for (const key of ["anyOf", "oneOf", "allOf"] as const) {
const branch = node[key];
if (Array.isArray(branch)) for (const sub of branch) collectSchemaTypes(sub, out, depth + 1);
}
return out;
}
export function jsonTypeOf(value: unknown): string {
const type = typeof value;
if (value === null) return "null";
if (type === "number" || type === "bigint") return "number";
if (type === "boolean") return "boolean";
if (type === "string") return "string";
return "object";
}
export function decodeValue(raw: string): unknown {
const trimmed = raw.trim();
if (trimmed.length === 0) return trimmed;
try {
return JSON.parse(trimmed) as unknown;
} catch {
return raw;
}
}
export function coerceValue(raw: string, schema: unknown): unknown {
return isStringOnlySchema(schema) ? raw : decodeValue(raw);
}
export function isArraySchema(schema: unknown): boolean {
return collectSchemaTypes(schema).has("array");
}
export function isObjectSchema(schema: unknown): boolean {
return collectSchemaTypes(schema).has("object");
}
export function getObjectProperties(schema: unknown): Record<string, unknown> {
if (!schema || typeof schema !== "object" || Array.isArray(schema)) return {};
const props = (schema as Record<string, unknown>).properties;
return props && typeof props === "object" && !Array.isArray(props) ? (props as Record<string, unknown>) : {};
}
export function getArrayItemSchema(schema: unknown): unknown {
if (!schema || typeof schema !== "object" || Array.isArray(schema)) return undefined;
return (schema as Record<string, unknown>).items;
}
let idCounter = 0;
export function mintToolCallId(): string {
idCounter = (idCounter + 1) % Number.MAX_SAFE_INTEGER;
return `ptc_${Date.now().toString(36)}_${idCounter.toString(36)}`;
}
export function partialSuffixOverlap(text: string, tag: string): number {
const max = Math.min(text.length, tag.length - 1);
for (let k = max; k > 0; k--) {
if (text.endsWith(tag.slice(0, k))) return k;
}
return 0;
}
export function partialSuffixOverlapAny(text: string, tags: readonly string[]): number {
let best = 0;
for (const tag of tags) best = Math.max(best, partialSuffixOverlap(text, tag));
return best;
}
export function normalizeKimiFunctionName(rawId: string): string {
const beforeIndex = rawId.split(":", 1)[0] ?? rawId;
const parts = beforeIndex.split(".");
return parts[parts.length - 1]?.trim() ?? beforeIndex.trim();
}
export function asRecord(value: unknown): Record<string, unknown> {
return value && typeof value === "object" && !Array.isArray(value) ? (value as Record<string, unknown>) : {};
}
+23
View File
@@ -0,0 +1,23 @@
## Format guide
A tool call wraps the function name, a separator, and one JSON object of arguments in fixed tokens. Emit them exactly:
```text
<|tool▁calls▁begin|><|tool▁call▁begin|>tool_name<|tool▁sep|>{"arg":"value"}<|tool▁call▁end|><|tool▁calls▁end|>
```
Results arrive as output tokens:
```text
<|tool▁output▁begin|>verbatim tool result<|tool▁output▁end|>
```
## Rules
- Use `|` (U+FF5C) and `▁` (U+2581) exactly.
- Tool name MUST match an available function; arguments are one valid JSON object.
- NEVER wrap arguments in Markdown fences; NEVER emit a `type` field or `function` prefix.
- Multiple calls chain `<|tool▁call▁begin|>...<|tool▁call▁end|>` directly — no separators, spaces, or newlines between them.
- Private reasoning, when needed, goes in `<think>...</think>` before the tokens.
- Read each output token in call order. NEVER emit output tokens yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+535
View File
@@ -0,0 +1,535 @@
import { parseJsonWithRepair } from "../utils/json-parse";
import { asRecord, mintToolCallId, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./deepseek.md" with { type: "text" };
import { renderDeepSeekInvocation, renderDeepSeekToolCalls, renderDeepSeekToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
export const DEEPSEEK_TOOL_CALLS_BEGIN = "<|tool▁calls▁begin|>";
export const DEEPSEEK_TOOL_CALLS_END = "<|tool▁calls▁end|>";
export const DEEPSEEK_TOOL_CALL_BEGIN = "<|tool▁call▁begin|>";
export const DEEPSEEK_TOOL_CALL_END = "<|tool▁call▁end|>";
export const DEEPSEEK_TOOL_SEPARATOR = "<|tool▁sep|>";
const THINK_OPEN = "<think>";
const THINK_CLOSE = "</think>";
const LEGACY_TOOL_TYPE = "function";
const LEGACY_JSON_FENCE = "```json";
const CODE_FENCE = "```";
const DSML_TOOL_CALLS_OPEN_FULLWIDTH = "<|DSML|tool_calls>";
const DSML_TOOL_CALLS_CLOSE_FULLWIDTH = "</|DSML|tool_calls>";
const DSML_TOOL_CALLS_OPEN_ASCII = "<|DSML|tool_calls>";
const DSML_TOOL_CALLS_CLOSE_ASCII = "</|DSML|tool_calls>";
const CONTROL_TOKENS = [
"<|begin▁of▁sentence|>",
"<|end▁of▁sentence|>",
"<|▁pad▁|>",
"<|User|>",
"<|Assistant|>",
"<|EOT|>",
"<|search▁begin|>",
"<|search▁end|>",
"<|fim▁hole|>",
"<|fim▁begin|>",
"<|fim▁end|>",
"<|tool▁outputs▁begin|>",
"<|tool▁outputs▁end|>",
"<|tool▁output▁begin|>",
"<|tool▁output▁end|>",
] as const;
const OUTSIDE_TOKENS = [
DEEPSEEK_TOOL_CALLS_BEGIN,
DEEPSEEK_TOOL_CALLS_END,
DEEPSEEK_TOOL_CALL_BEGIN,
THINK_OPEN,
THINK_CLOSE,
DSML_TOOL_CALLS_OPEN_FULLWIDTH,
DSML_TOOL_CALLS_OPEN_ASCII,
DSML_TOOL_CALLS_CLOSE_FULLWIDTH,
DSML_TOOL_CALLS_CLOSE_ASCII,
...CONTROL_TOKENS,
] as const;
const SECTION_TOKENS = [DEEPSEEK_TOOL_CALL_BEGIN, DEEPSEEK_TOOL_CALLS_END] as const;
const DSML_SECTION_TOKENS = [
DSML_TOOL_CALLS_CLOSE_FULLWIDTH,
DSML_TOOL_CALLS_CLOSE_ASCII,
"<|DSML|invoke",
"<|DSML|invoke",
] as const;
const DSML_INVOKE_TOKENS = ["</|DSML|invoke>", "</|DSML|invoke>", "<|DSML|parameter", "<|DSML|parameter"] as const;
const DSML_PARAMETER_CLOSE_TOKENS = ["</|DSML|parameter>", "</|DSML|parameter>"] as const;
type State =
| "outside"
| "thinking"
| "section"
| "header"
| "args"
| "legacyName"
| "legacyArgs"
| "dsmlSection"
| "dsmlInvoke"
| "dsmlParam";
export class DeepSeekInbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
#parseThinking: boolean;
#inToolSection = false;
#id = "";
#name = "";
#thinking = "";
#dsmlArgs: Record<string, unknown> = {};
#dsmlParamName = "";
#dsmlParamIsString = true;
#rawBlock = "";
#stripLeadingWhitespace = false;
constructor(options: InbandScannerOptions = {}) {
this.#parseThinking = options.parseThinking ?? true;
}
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
this.#consumeOutside(final, events);
if (this.#state !== "outside" && this.#buffer.length > 0) continue;
break;
}
if (this.#state === "thinking") {
this.#consumeThinking(final, events);
if (!final && this.#state === "thinking") break;
continue;
}
if (this.#state === "section") {
if (!this.#consumeSection(final)) break;
continue;
}
if (this.#state === "header") {
if (!this.#consumeHeader(final, events)) break;
continue;
}
if (this.#state === "legacyName") {
if (!this.#consumeLegacyName(final, events)) break;
continue;
}
if (this.#state === "args" || this.#state === "legacyArgs") {
if (!this.#consumeArgs(final, events)) break;
continue;
}
if (this.#state === "dsmlSection") {
if (!this.#consumeDsmlSection(final, events)) break;
continue;
}
if (this.#state === "dsmlInvoke") {
if (!this.#consumeDsmlInvoke(final, events)) break;
continue;
}
if (!this.#consumeDsmlParam(final)) break;
}
if (final && this.#buffer.length === 0 && this.#rawBlock.length > 0) this.#rawBlock = "";
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): void {
while (this.#buffer.length > 0) {
if (this.#stripLeadingWhitespace) {
// A chat-template control token (e.g. `<|Assistant|>`) was just dropped;
// swallow the template whitespace that trails it so it never leaks into
// visible text. Whitespace can't begin another token, so eager trim is safe.
const trimmed = this.#buffer.replace(/^\s+/u, "");
if (trimmed.length === 0) {
this.#buffer = "";
return;
}
this.#buffer = trimmed;
this.#stripLeadingWhitespace = false;
}
const match = findEarliestToken(this.#buffer, OUTSIDE_TOKENS);
if (!match) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OUTSIDE_TOKENS);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
return;
}
if (match.index > 0) events.push({ type: "text", text: this.#buffer.slice(0, match.index) });
this.#buffer = this.#buffer.slice(match.index);
if (this.#buffer.startsWith(DEEPSEEK_TOOL_CALLS_BEGIN)) {
this.#buffer = this.#buffer.slice(DEEPSEEK_TOOL_CALLS_BEGIN.length);
this.#inToolSection = true;
this.#state = "section";
return;
}
if (this.#buffer.startsWith(DEEPSEEK_TOOL_CALL_BEGIN)) {
this.#buffer = this.#buffer.slice(DEEPSEEK_TOOL_CALL_BEGIN.length);
this.#rawBlock = DEEPSEEK_TOOL_CALL_BEGIN;
this.#inToolSection = false;
this.#state = "header";
return;
}
if (this.#buffer.startsWith(THINK_OPEN)) {
this.#buffer = this.#buffer.slice(THINK_OPEN.length);
this.#state = "thinking";
this.#thinking = "";
if (this.#parseThinking) events.push({ type: "thinkingStart" });
return;
}
if (
this.#buffer.startsWith(DSML_TOOL_CALLS_OPEN_FULLWIDTH) ||
this.#buffer.startsWith(DSML_TOOL_CALLS_OPEN_ASCII)
) {
const openToken = this.#buffer.startsWith(DSML_TOOL_CALLS_OPEN_FULLWIDTH)
? DSML_TOOL_CALLS_OPEN_FULLWIDTH
: DSML_TOOL_CALLS_OPEN_ASCII;
this.#buffer = this.#buffer.slice(openToken.length);
this.#state = "dsmlSection";
return;
}
const control = this.#matchingControlToken();
if (control) {
this.#buffer = this.#buffer.slice(control.length);
this.#stripLeadingWhitespace = true;
continue;
}
this.#buffer = this.#buffer.slice(match.token.length);
}
}
#consumeThinking(final: boolean, events: InbandScanEvent[]): void {
const close = this.#buffer.indexOf(THINK_CLOSE);
if (close === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, [THINK_CLOSE]);
this.#emitThinking(this.#buffer.slice(0, this.#buffer.length - hold), events);
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
if (final) this.#endThinking(events);
return;
}
this.#emitThinking(this.#buffer.slice(0, close), events);
this.#buffer = this.#buffer.slice(close + THINK_CLOSE.length);
this.#endThinking(events);
}
#consumeSection(final: boolean): boolean {
while (this.#buffer.length > 0) {
this.#skipWhitespace();
if (this.#buffer.startsWith(DEEPSEEK_TOOL_CALLS_END)) {
this.#buffer = this.#buffer.slice(DEEPSEEK_TOOL_CALLS_END.length);
this.#inToolSection = false;
this.#state = "outside";
return true;
}
if (this.#buffer.startsWith(DEEPSEEK_TOOL_CALL_BEGIN)) {
this.#buffer = this.#buffer.slice(DEEPSEEK_TOOL_CALL_BEGIN.length);
this.#rawBlock = DEEPSEEK_TOOL_CALL_BEGIN;
this.#state = "header";
return true;
}
if (!final && partialSuffixOverlapAny(this.#buffer, SECTION_TOKENS) === this.#buffer.length) return false;
if (this.#buffer.length === 0) return false;
this.#buffer = this.#buffer.slice(1);
}
return final;
}
#consumeHeader(final: boolean, events: InbandScanEvent[]): boolean {
const sep = this.#buffer.indexOf(DEEPSEEK_TOOL_SEPARATOR);
if (sep === -1) {
if (final) this.#resetTool();
return false;
}
const rawHead = this.#buffer.slice(0, sep + DEEPSEEK_TOOL_SEPARATOR.length);
const head = this.#buffer.slice(0, sep).trim();
this.#rawBlock += rawHead;
this.#buffer = this.#buffer.slice(rawHead.length);
if (head === LEGACY_TOOL_TYPE) {
this.#state = "legacyName";
return true;
}
this.#startTool(head, events);
this.#state = "args";
return true;
}
#consumeLegacyName(final: boolean, events: InbandScanEvent[]): boolean {
const fence = this.#buffer.indexOf(LEGACY_JSON_FENCE);
if (fence === -1) {
if (final) this.#resetTool();
return false;
}
const rawName = this.#buffer.slice(0, fence + LEGACY_JSON_FENCE.length);
const name = this.#buffer.slice(0, fence).trim();
this.#rawBlock += rawName;
this.#buffer = this.#buffer.slice(rawName.length);
this.#rawBlock += this.#dropOneLineBreak();
this.#startTool(name, events);
this.#state = "legacyArgs";
return true;
}
#consumeArgs(final: boolean, events: InbandScanEvent[]): boolean {
const end = this.#buffer.indexOf(DEEPSEEK_TOOL_CALL_END);
if (end === -1) {
if (final) this.#resetTool();
return false;
}
let rawArgs = this.#buffer.slice(0, end);
if (this.#state === "legacyArgs") {
const fence = rawArgs.lastIndexOf(CODE_FENCE);
if (fence !== -1) rawArgs = rawArgs.slice(0, fence);
}
const rawTail = this.#buffer.slice(0, end + DEEPSEEK_TOOL_CALL_END.length);
this.#rawBlock += rawTail;
events.push({
type: "toolEnd",
id: this.#id,
name: this.#name,
arguments: this.#parseArgs(rawArgs),
rawBlock: this.#rawBlock,
});
this.#buffer = this.#buffer.slice(rawTail.length);
this.#resetTool(this.#inToolSection ? "section" : "outside");
return true;
}
#consumeDsmlSection(final: boolean, events: InbandScanEvent[]): boolean {
while (this.#buffer.length > 0) {
this.#skipWhitespace();
const close = this.#matchingDsmlClose(DSML_TOOL_CALLS_CLOSE_FULLWIDTH, DSML_TOOL_CALLS_CLOSE_ASCII);
if (close) {
this.#buffer = this.#buffer.slice(close.length);
this.#state = "outside";
return true;
}
const invoke = this.#matchDsmlOpen("invoke");
if (invoke) {
this.#rawBlock = invoke.raw;
this.#name = invoke.name;
this.#id = mintToolCallId();
this.#dsmlArgs = {};
events.push({ type: "toolStart", id: this.#id, name: this.#name });
this.#state = "dsmlInvoke";
return true;
}
if (!final) {
if (
(this.#buffer.startsWith("<|DSML|invoke") || this.#buffer.startsWith("<|DSML|invoke")) &&
!this.#buffer.includes(">")
)
return false;
if (partialSuffixOverlapAny(this.#buffer, DSML_SECTION_TOKENS) === this.#buffer.length) return false;
}
if (this.#buffer.length === 0) return false;
this.#buffer = this.#buffer.slice(1);
}
return final;
}
#consumeDsmlInvoke(final: boolean, events: InbandScanEvent[]): boolean {
while (this.#buffer.length > 0) {
const skipped = this.#skipWhitespace();
if (skipped.length > 0) this.#rawBlock += skipped;
const close = this.#matchingDsmlClose("</|DSML|invoke>", "</|DSML|invoke>");
if (close) {
this.#rawBlock += close;
this.#buffer = this.#buffer.slice(close.length);
events.push({
type: "toolEnd",
id: this.#id,
name: this.#name,
arguments: this.#dsmlArgs,
rawBlock: this.#rawBlock,
});
this.#resetDsmlTool();
this.#state = "dsmlSection";
return true;
}
const param = this.#matchDsmlOpen("parameter");
if (param) {
this.#rawBlock += param.raw;
this.#dsmlParamName = param.name;
this.#dsmlParamIsString = param.stringAttr !== "false";
this.#state = "dsmlParam";
return true;
}
if (!final) {
if (
(this.#buffer.startsWith("<|DSML|parameter") || this.#buffer.startsWith("<|DSML|parameter")) &&
!this.#buffer.includes(">")
)
return false;
if (partialSuffixOverlapAny(this.#buffer, DSML_INVOKE_TOKENS) === this.#buffer.length) return false;
}
const consumed = this.#buffer[0]!;
this.#rawBlock += consumed;
this.#buffer = this.#buffer.slice(1);
}
return final;
}
#consumeDsmlParam(final: boolean): boolean {
const close = findEarliestToken(this.#buffer, DSML_PARAMETER_CLOSE_TOKENS);
if (!close) {
if (final) this.#resetDsmlTool();
return false;
}
const rawValue = this.#buffer.slice(0, close.index);
this.#dsmlArgs[this.#dsmlParamName] = coerceDsmlValue(rawValue, this.#dsmlParamIsString);
this.#rawBlock += rawValue + close.token;
this.#buffer = this.#buffer.slice(close.index + close.token.length);
this.#dsmlParamName = "";
this.#dsmlParamIsString = true;
this.#state = "dsmlInvoke";
return true;
}
#startTool(name: string, events: InbandScanEvent[]): void {
this.#name = name;
this.#id = mintToolCallId();
events.push({ type: "toolStart", id: this.#id, name: this.#name });
}
#emitThinking(delta: string, events: InbandScanEvent[]): void {
if (delta.length === 0) return;
if (this.#parseThinking) {
this.#thinking += delta;
events.push({ type: "thinkingDelta", delta });
} else {
events.push({ type: "text", text: delta });
}
}
#endThinking(events: InbandScanEvent[]): void {
if (this.#parseThinking) events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#state = "outside";
}
#parseArgs(rawArgs: string): Record<string, unknown> {
const trimmed = rawArgs.trim();
if (trimmed.length === 0) return {};
try {
return asRecord(parseJsonWithRepair<unknown>(trimmed));
} catch {
return {};
}
}
#skipWhitespace(): string {
let i = 0;
while (i < this.#buffer.length && /\s/.test(this.#buffer[i]!)) i++;
if (i === 0) return "";
const skipped = this.#buffer.slice(0, i);
this.#buffer = this.#buffer.slice(i);
return skipped;
}
#dropOneLineBreak(): string {
if (this.#buffer.startsWith("\r\n")) {
this.#buffer = this.#buffer.slice(2);
return "\r\n";
}
if (this.#buffer.startsWith("\n")) {
this.#buffer = this.#buffer.slice(1);
return "\n";
}
return "";
}
#matchingControlToken(): string | undefined {
if (this.#buffer.startsWith(DEEPSEEK_TOOL_CALLS_END)) return DEEPSEEK_TOOL_CALLS_END;
if (this.#buffer.startsWith(THINK_CLOSE)) return THINK_CLOSE;
if (this.#buffer.startsWith(DSML_TOOL_CALLS_CLOSE_FULLWIDTH)) return DSML_TOOL_CALLS_CLOSE_FULLWIDTH;
if (this.#buffer.startsWith(DSML_TOOL_CALLS_CLOSE_ASCII)) return DSML_TOOL_CALLS_CLOSE_ASCII;
for (const token of CONTROL_TOKENS) {
if (this.#buffer.startsWith(token)) return token;
}
return undefined;
}
#matchingDsmlClose(fullwidth: string, ascii: string): string | undefined {
if (this.#buffer.startsWith(fullwidth)) return fullwidth;
if (this.#buffer.startsWith(ascii)) return ascii;
return undefined;
}
#matchDsmlOpen(
kind: "invoke" | "parameter",
): { name: string; stringAttr: string | undefined; raw: string } | undefined {
if (!this.#buffer.startsWith(`<|DSML|${kind}`) && !this.#buffer.startsWith(`<|DSML|${kind}`)) return undefined;
const end = this.#buffer.indexOf(">");
if (end === -1) return undefined;
const tag = this.#buffer.slice(0, end + 1);
const name = /\sname="([^"]*)"/.exec(tag)?.[1];
if (name === undefined) return undefined;
const stringAttr = /\sstring="(true|false)"/.exec(tag)?.[1];
this.#buffer = this.#buffer.slice(end + 1);
return { name, stringAttr, raw: tag };
}
#resetTool(next: State = "outside"): void {
this.#state = next;
this.#id = "";
this.#name = "";
this.#rawBlock = "";
}
#resetDsmlTool(): void {
this.#id = "";
this.#name = "";
this.#dsmlArgs = {};
this.#dsmlParamName = "";
this.#dsmlParamIsString = true;
this.#rawBlock = "";
}
}
function findEarliestToken(text: string, tokens: readonly string[]): { index: number; token: string } | undefined {
let bestIndex = -1;
let bestToken = "";
for (const token of tokens) {
const index = text.indexOf(token);
if (index === -1) continue;
if (bestIndex === -1 || index < bestIndex || (index === bestIndex && token.length > bestToken.length)) {
bestIndex = index;
bestToken = token;
}
}
return bestIndex === -1 ? undefined : { index: bestIndex, token: bestToken };
}
function coerceDsmlValue(raw: string, isString: boolean): unknown {
if (isString) return raw;
const trimmed = raw.trim();
if (trimmed.length === 0) return raw;
try {
return parseJsonWithRepair<unknown>(trimmed);
} catch {
return raw;
}
}
const grammar: Grammar = {
syntax: "deepseek",
prompt: grammarPrompt,
createScanner: options => new DeepSeekInbandScanner(options),
renderToolCall: renderDeepSeekInvocation,
renderAssistantToolCalls: renderDeepSeekToolCalls,
renderToolResults: renderDeepSeekToolResults,
};
export default grammar;
+33
View File
@@ -0,0 +1,33 @@
import type { ToolCall } from "../types";
import { getInbandGrammar } from "./factory";
import type { InbandTool, ToolCallSyntax } from "./types";
const INTENT_PLACEHOLDER = "…";
export function renderToolExamples(tool: InbandTool, syntax: ToolCallSyntax, intentField?: string): string {
const examples = tool.examples;
if (!examples?.length) return "";
const grammar = getInbandGrammar(syntax);
const renderCall = (args: Record<string, unknown>): string => {
// When intent tracing injects `_i` into the schema, examples must show a
// placeholder so the model learns to emit it. Keep it first, matching the
// schema injection order.
const finalArgs = intentField ? { [intentField]: INTENT_PLACEHOLDER, ...args } : args;
const call: ToolCall = {
type: "toolCall",
id: "example",
name: tool.name,
arguments: finalArgs,
};
return `<example>\n${grammar.renderToolCall(call, { tools: [tool], example: true }).trim()}\n</example>`;
};
const parts = examples.map(ex => {
const head = ex.caption ? `# ${ex.caption}\n` : "";
if ("call" in ex) return head + renderCall(ex.call);
if ("good" in ex) {
return `${head}WRONG:\n${renderCall(ex.bad)}\nRIGHT:\n${renderCall(ex.good)}`;
}
return head.trimEnd() + (ex.note ? `\n${ex.note}` : "");
});
return `<examples>\n${parts.join("\n")}\n</examples>`;
}
+34
View File
@@ -0,0 +1,34 @@
import anthropicGrammar from "./anthropic";
import deepseekGrammar from "./deepseek";
import geminiGrammar from "./gemini";
import gemmaGrammar from "./gemma";
import glmGrammar from "./glm";
import harmonyGrammar from "./harmony";
import hermesGrammar from "./hermes";
import kimiGrammar from "./kimi";
import piGrammar from "./pi";
import qwen3Grammar from "./qwen3";
import type { Grammar, InbandScanner, InbandScannerOptions, ToolCallSyntax } from "./types";
import xmlGrammar from "./xml";
const GRAMMARS: Record<ToolCallSyntax, Grammar> = {
glm: glmGrammar,
hermes: hermesGrammar,
kimi: kimiGrammar,
xml: xmlGrammar,
anthropic: anthropicGrammar,
deepseek: deepseekGrammar,
harmony: harmonyGrammar,
pi: piGrammar,
qwen3: qwen3Grammar,
gemini: geminiGrammar,
gemma: gemmaGrammar,
};
export function getInbandGrammar(syntax: ToolCallSyntax): Grammar {
return GRAMMARS[syntax];
}
export function createInbandScanner(syntax: ToolCallSyntax, options: InbandScannerOptions = {}): InbandScanner {
return getInbandGrammar(syntax).createScanner(options);
}
+35
View File
@@ -0,0 +1,35 @@
## Format guide
Emit tool calls as Python inside a fenced ` ```tool_code ` block. Call each function as a method on `default_api`:
````text
```tool_code
default_api.function_name(arg="value", count=2)
```
````
Argument values are Python literals: `"strings"`, numbers, `True`/`False`, `None`, `[lists]`, `{"dicts": 1}`.
Call several functions in parallel as a Python list:
````text
```tool_code
[default_api.first(x="a"), default_api.second(y="b")]
```
````
Tool results arrive later in a ` ```tool_outputs ` block:
````text
```tool_outputs
verbatim tool result
```
````
## Rules
- The function name MUST match a listed function; arguments are keyword form (`name=value`).
- Multiple calls = a single `[...]` list (or one `default_api...` call per line) inside one ` ```tool_code ` block.
- Put any reasoning as plain text before the ` ```tool_code ` block, never inside it.
- Read each ` ```tool_outputs ` block in call order. NEVER write a ` ```tool_outputs ` block yourself.
- After emitting the ` ```tool_code ` block, YOU MUST STOP AND HALT.
+440
View File
@@ -0,0 +1,440 @@
import { mintToolCallId, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./gemini.md" with { type: "text" };
import { renderGeminiInvocation, renderGeminiToolCalls, renderGeminiToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner } from "./types";
const CODE_OPEN = "```tool_code";
const FENCE = "```";
const OPEN_TAGS = [CODE_OPEN] as const;
type State = "outside" | "tool";
interface ParsedCall {
name: string;
arguments: Record<string, unknown>;
}
/**
* Scanner for the hosted-Gemini / Gemma 3 Pythonic tool-calling convention
* (see `docs/toolconv/gemini.md`). Tool calls arrive as a ```` ```tool_code ````
* fenced block whose body is one or more Python call expressions, e.g.
* `print(default_api.search(pattern="x", skip=40))`. Like the qwen3 scanner we
* buffer the whole block until its closing fence, then parse all calls at once
* (no incremental argument deltas — Python literals are not worth streaming).
*/
export class GeminiInbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
this.#consumeOutside(final, events);
if (this.#state === "outside") break;
continue;
}
this.#consumeTool(final, events);
if (this.#state === "tool") break;
}
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): void {
const open = this.#buffer.indexOf(CODE_OPEN);
if (open === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OPEN_TAGS);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
return;
}
if (open > 0) events.push({ type: "text", text: this.#buffer.slice(0, open) });
this.#buffer = this.#buffer.slice(open + CODE_OPEN.length);
this.#state = "tool";
}
#consumeTool(final: boolean, events: InbandScanEvent[]): void {
const close = this.#buffer.indexOf(FENCE);
if (close === -1) {
// Inside the fence we emit nothing until it closes; on a truncated
// stream the incomplete block is dropped rather than leaked as text.
if (final) {
this.#buffer = "";
this.#state = "outside";
}
return;
}
const body = this.#buffer.slice(0, close);
const rawBlock = `${CODE_OPEN}${body}${FENCE}`;
for (const call of parseGeminiCalls(body)) {
const id = mintToolCallId();
events.push({ type: "toolStart", id, name: call.name });
events.push({ type: "toolEnd", id, name: call.name, arguments: call.arguments, rawBlock });
}
this.#buffer = this.#buffer.slice(close + FENCE.length);
this.#state = "outside";
}
}
/** Extract every top-level call expression in a `tool_code` body. */
function parseGeminiCalls(body: string): ParsedCall[] {
const calls: ParsedCall[] = [];
let i = 0;
const n = body.length;
while (i < n) {
const ch = body[i]!;
if (ch === '"' || ch === "'") {
i = skipString(body, i);
continue;
}
if (ch === "#") {
i = skipComment(body, i);
continue;
}
if (ch === "(") {
const name = identBefore(body, i);
if (name && name !== "print") {
const end = matchParen(body, i);
if (end !== -1) {
calls.push({ name, arguments: parsePyArgs(body.slice(i + 1, end)) });
i = end + 1;
continue;
}
}
}
i++;
}
return calls;
}
/** Identifier immediately preceding a `(` (the callee's final name segment). */
function identBefore(body: string, parenIndex: number): string | undefined {
let j = parenIndex - 1;
while (j >= 0 && /\s/.test(body[j]!)) j--;
const end = j + 1;
while (j >= 0 && /[A-Za-z0-9_]/.test(body[j]!)) j--;
const name = body.slice(j + 1, end);
return /^[A-Za-z_]\w*$/.test(name) ? name : undefined;
}
/** Index of the `)` matching the `(` at `openIndex`, skipping string contents. */
function matchParen(body: string, openIndex: number): number {
let depth = 0;
let i = openIndex;
const n = body.length;
while (i < n) {
const ch = body[i]!;
if (ch === '"' || ch === "'") {
i = skipString(body, i);
continue;
}
if (ch === "#") {
i = skipComment(body, i);
continue;
}
if (ch === "(") depth++;
else if (ch === ")" && --depth === 0) return i;
i++;
}
return -1;
}
/** Index just past the Python string literal starting at `i` (a quote char). */
function skipString(body: string, i: number): number {
const quote = body[i]!;
const triple = quote + quote + quote;
if (body.startsWith(triple, i)) {
const close = body.indexOf(triple, i + 3);
return close === -1 ? body.length : close + 3;
}
let j = i + 1;
const n = body.length;
while (j < n) {
const ch = body[j]!;
if (ch === "\\") {
j += 2;
continue;
}
if (ch === quote) return j + 1;
j++;
}
return n;
}
function skipComment(body: string, i: number): number {
const newline = body.indexOf("\n", i + 1);
return newline === -1 ? body.length : newline + 1;
}
function stripComments(body: string): string {
let out = "";
let i = 0;
const n = body.length;
while (i < n) {
const ch = body[i]!;
if (ch === '"' || ch === "'") {
const end = skipString(body, i);
out += body.slice(i, end);
i = end;
continue;
}
if (ch === "#") {
const newline = body.indexOf("\n", i + 1);
if (newline === -1) break;
out += "\n";
i = newline + 1;
continue;
}
out += ch;
i++;
}
return out;
}
function parsePyArgs(text: string): Record<string, unknown> {
const out: Record<string, unknown> = {};
for (const segment of splitTopLevel(stripComments(text), ",")) {
const trimmed = segment.trim();
if (trimmed.length === 0) continue;
const eq = topLevelIndexOf(trimmed, "=");
if (eq === -1) continue; // positional args are not part of the convention
const key = trimmed.slice(0, eq).trim();
if (!/^[A-Za-z_]\w*$/.test(key)) continue;
out[key] = parsePyValue(trimmed.slice(eq + 1).trim());
}
return out;
}
function parsePyValue(raw: string): unknown {
const t = raw.trim();
if (t.length === 0) return "";
if (t === "True" || t === "true") return true;
if (t === "False" || t === "false") return false;
if (t === "None" || t === "null") return null;
const prefix = stringPrefixLength(t);
if (prefix !== undefined) return decodeString(t);
const first = t[0]!;
if (first === "[") return parseList(t);
if (first === "{") return parseDict(t);
if (/^[+-]?(\d|\.)/.test(t)) {
const num = Number(t);
if (!Number.isNaN(num)) return num;
}
return t;
}
function parseList(t: string): unknown[] {
const inner = t.slice(1, t.endsWith("]") ? t.length - 1 : t.length);
return splitTopLevel(stripComments(inner), ",")
.map(part => part.trim())
.filter(part => part.length > 0)
.map(parsePyValue);
}
function parseDict(t: string): Record<string, unknown> {
const inner = t.slice(1, t.endsWith("}") ? t.length - 1 : t.length);
const out: Record<string, unknown> = {};
for (const segment of splitTopLevel(stripComments(inner), ",")) {
const trimmed = segment.trim();
if (trimmed.length === 0) continue;
const colon = topLevelIndexOf(trimmed, ":");
if (colon === -1) continue;
const keyRaw = trimmed.slice(0, colon).trim();
const key = stringPrefixLength(keyRaw) !== undefined ? decodeString(keyRaw) : keyRaw;
out[key] = parsePyValue(trimmed.slice(colon + 1).trim());
}
return out;
}
function decodeString(t: string): string {
const prefix = stringPrefixLength(t) ?? 0;
const raw = t.slice(0, prefix).toLowerCase().includes("r");
const quote = t[prefix]!;
const triple = quote + quote + quote;
if (t.startsWith(triple, prefix) && t.length >= prefix + 6 && t.endsWith(triple)) {
const inner = t.slice(prefix + 3, t.length - 3);
return raw ? inner : unescapePythonString(inner);
}
const inner = t.endsWith(quote) && t.length >= prefix + 2 ? t.slice(prefix + 1, t.length - 1) : t.slice(prefix + 1);
return raw ? inner : unescapePythonString(inner);
}
function stringPrefixLength(t: string): number | undefined {
for (const len of [2, 1, 0]) {
const prefix = t.slice(0, len).toLowerCase();
if (
(prefix === "" || prefix === "r" || prefix === "u" || prefix === "b" || prefix === "br" || prefix === "rb") &&
(t[len] === '"' || t[len] === "'")
) {
return len;
}
}
return undefined;
}
function unescapePythonString(s: string): string {
if (!s.includes("\\")) return s;
let out = "";
let i = 0;
while (i < s.length) {
const ch = s[i]!;
if (ch !== "\\") {
out += ch;
i++;
continue;
}
const next = s[i + 1];
if (next && /^[0-7]$/.test(next)) {
const octal = /^[0-7]{1,3}/.exec(s.slice(i + 1))![0];
out += String.fromCharCode(parseInt(octal, 8));
i += octal.length + 1;
continue;
}
switch (next) {
case "n":
out += "\n";
i += 2;
break;
case "t":
out += "\t";
i += 2;
break;
case "r":
out += "\r";
i += 2;
break;
case "\\":
out += "\\";
i += 2;
break;
case "'":
out += "'";
i += 2;
break;
case '"':
out += '"';
i += 2;
break;
case "0":
out += "\0";
i += 2;
break;
case "x": {
const hex = s.slice(i + 2, i + 4);
if (/^[0-9a-fA-F]{2}$/.test(hex)) {
out += String.fromCharCode(parseInt(hex, 16));
i += 4;
} else {
out += "x";
i += 2;
}
break;
}
case "u": {
const hex = s.slice(i + 2, i + 6);
if (/^[0-9a-fA-F]{4}$/.test(hex)) {
out += String.fromCharCode(parseInt(hex, 16));
i += 6;
} else {
out += "u";
i += 2;
}
break;
}
case "U": {
const hex = s.slice(i + 2, i + 10);
if (/^[0-9a-fA-F]{8}$/.test(hex)) {
out += String.fromCodePoint(parseInt(hex, 16));
i += 10;
} else {
out += "U";
i += 2;
}
break;
}
case undefined:
out += "\\";
i += 1;
break;
default:
out += next;
i += 2;
break;
}
}
return out;
}
/** Split on `sep` at bracket depth 0, skipping string literals. */
function splitTopLevel(text: string, sep: string): string[] {
const parts: string[] = [];
let depth = 0;
let start = 0;
let i = 0;
const n = text.length;
while (i < n) {
const ch = text[i]!;
if (ch === '"' || ch === "'") {
i = skipString(text, i);
continue;
}
if (ch === "#") {
i = skipComment(text, i);
continue;
}
if (ch === "(" || ch === "[" || ch === "{") depth++;
else if (ch === ")" || ch === "]" || ch === "}") depth--;
else if (depth === 0 && ch === sep) {
parts.push(text.slice(start, i));
start = i + 1;
}
i++;
}
parts.push(text.slice(start));
return parts;
}
/** First index of `ch` at bracket depth 0, skipping string literals. */
function topLevelIndexOf(text: string, ch: string): number {
let depth = 0;
let i = 0;
const n = text.length;
while (i < n) {
const c = text[i]!;
if (c === '"' || c === "'") {
i = skipString(text, i);
continue;
}
if (c === "#") {
i = skipComment(text, i);
continue;
}
if (c === "(" || c === "[" || c === "{") depth++;
else if (c === ")" || c === "]" || c === "}") depth--;
else if (depth === 0 && c === ch) return i;
i++;
}
return -1;
}
const grammar: Grammar = {
syntax: "gemini",
prompt: grammarPrompt,
createScanner: () => new GeminiInbandScanner(),
renderToolCall: renderGeminiInvocation,
renderAssistantToolCalls: renderGeminiToolCalls,
renderToolResults: renderGeminiToolResults,
};
export default grammar;
+23
View File
@@ -0,0 +1,23 @@
## Format guide
Emit each tool call as one `<|tool_call>` block. The body is `call:NAME{key:value,...}`; wrap every string value in the `<|"|>` token:
```text
<|tool_call>call:function_name{path:<|"|>src/a.ts<|"|>,count:2}<tool_call|>
```
Non-string values are bare: numbers (`2`), `true`/`false`, `null`, lists `[<|"|>a<|"|>,<|"|>b<|"|>]`, and nested objects `{k:<|"|>v<|"|>}`.
Tool results arrive later in matching `<|tool_response>` blocks:
```text
<|tool_response>response:function_name{output:<|"|>verbatim result<|"|>}<tool_response|>
```
## Rules
- `NAME` MUST match a listed function; arguments are `key:value` pairs separated by commas.
- Multiple calls = consecutive `<|tool_call>...<tool_call|>` blocks; keep prose outside them.
- The closer is `<tool_call|>` (pipe on the right), not `</tool_call>` or `<|tool_call>`.
- Read each `<|tool_response>` block in call order. NEVER write a `<|tool_response>` block yourself.
- After emitting your tool calls, YOU MUST STOP AND HALT.
+237
View File
@@ -0,0 +1,237 @@
import { mintToolCallId, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./gemma.md" with { type: "text" };
import { renderGemmaInvocation, renderGemmaToolCalls, renderGemmaToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner } from "./types";
const CALL_OPEN = "<|tool_call>";
const CALL_CLOSE = "<tool_call|>";
const STRING = '<|"|>';
const OPEN_TAGS = [CALL_OPEN] as const;
const CALL_HEAD = /^call:\s*([A-Za-z_]\w*)\s*\{/;
type State = "outside" | "tool";
interface ParsedCall {
name: string;
arguments: Record<string, unknown>;
}
/**
* Scanner for the Gemma 4 token-delimited tool-calling convention (see
* `docs/toolconv/gemma.md`). Each call is one `<|tool_call>call:NAME{…}<tool_call|>`
* block whose argument list is `key:value` pairs; string values are wrapped in
* the `<|"|>` token rather than ASCII quotes, so splitting must skip those spans.
*/
export class GemmaInbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
this.#consumeOutside(final, events);
if (this.#state === "outside") break;
continue;
}
this.#consumeTool(final, events);
if (this.#state === "tool") break;
}
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): void {
const open = this.#buffer.indexOf(CALL_OPEN);
if (open === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OPEN_TAGS);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
return;
}
if (open > 0) events.push({ type: "text", text: this.#buffer.slice(0, open) });
this.#buffer = this.#buffer.slice(open + CALL_OPEN.length);
this.#state = "tool";
}
#consumeTool(final: boolean, events: InbandScanEvent[]): void {
const close = findCallClose(this.#buffer);
if (close === -1) {
if (final) {
this.#buffer = "";
this.#state = "outside";
}
return;
}
const body = this.#buffer.slice(0, close);
const parsed = parseGemmaCall(body);
if (parsed) {
const id = mintToolCallId();
events.push({ type: "toolStart", id, name: parsed.name });
events.push({
type: "toolEnd",
id,
name: parsed.name,
arguments: parsed.arguments,
rawBlock: `${CALL_OPEN}${body}${CALL_CLOSE}`,
});
}
this.#buffer = this.#buffer.slice(close + CALL_CLOSE.length);
this.#state = "outside";
}
}
function parseGemmaCall(body: string): ParsedCall | undefined {
const trimmed = body.trim();
const head = CALL_HEAD.exec(trimmed);
if (!head) return undefined;
const braceStart = head[0].length - 1;
const end = matchDelim(trimmed, braceStart, "{", "}");
const argsText = end === -1 ? trimmed.slice(braceStart + 1) : trimmed.slice(braceStart + 1, end);
return { name: head[1]!, arguments: parseGemmaArgs(argsText) };
}
function parseGemmaArgs(text: string): Record<string, unknown> {
const out: Record<string, unknown> = {};
for (const segment of splitTopLevel(text, ",")) {
const trimmed = segment.trim();
if (trimmed.length === 0) continue;
const colon = topLevelIndexOf(trimmed, ":");
if (colon === -1) continue;
const key = trimmed.slice(0, colon).trim();
if (!/^[A-Za-z_]\w*$/.test(key)) continue;
out[key] = parseGemmaValue(trimmed.slice(colon + 1).trim());
}
return out;
}
function parseGemmaValue(raw: string): unknown {
const t = raw.trim();
if (t.startsWith(STRING)) {
const close = t.indexOf(STRING, STRING.length);
return close === -1 ? t.slice(STRING.length) : t.slice(STRING.length, close);
}
if (t.startsWith("[")) {
const end = matchDelim(t, 0, "[", "]");
const inner = end === -1 ? t.slice(1) : t.slice(1, end);
return splitTopLevel(inner, ",")
.map(part => part.trim())
.filter(part => part.length > 0)
.map(parseGemmaValue);
}
if (t.startsWith("{")) {
const end = matchDelim(t, 0, "{", "}");
return parseGemmaArgs(end === -1 ? t.slice(1) : t.slice(1, end));
}
if (t === "true") return true;
if (t === "false") return false;
if (t === "null" || t === "none" || t === "None") return null;
if (/^[+-]?(\d|\.)/.test(t)) {
const num = Number(t);
if (!Number.isNaN(num)) return num;
}
return t;
}
/** Index just past the `<|"|>`-delimited string starting at `i`. */
function skipGemmaString(text: string, i: number): number {
const close = text.indexOf(STRING, i + STRING.length);
return close === -1 ? text.length : close + STRING.length;
}
function findCallClose(text: string): number {
let i = 0;
const n = text.length;
while (i < n) {
if (text.startsWith(STRING, i)) {
i = skipGemmaString(text, i);
continue;
}
if (text.startsWith(CALL_CLOSE, i)) return i;
i++;
}
return -1;
}
/** Index of the `close` delimiter matching `open` at `openIndex`, skipping strings. */
function matchDelim(text: string, openIndex: number, open: string, close: string): number {
let depth = 0;
let i = openIndex;
const n = text.length;
while (i < n) {
if (text.startsWith(STRING, i)) {
i = skipGemmaString(text, i);
continue;
}
const ch = text[i]!;
if (ch === open) depth++;
else if (ch === close && --depth === 0) return i;
i++;
}
return -1;
}
/** Split on `sep` at bracket depth 0, skipping `<|"|>` string spans. */
function splitTopLevel(text: string, sep: string): string[] {
const parts: string[] = [];
let depth = 0;
let start = 0;
let i = 0;
const n = text.length;
while (i < n) {
if (text.startsWith(STRING, i)) {
i = skipGemmaString(text, i);
continue;
}
const ch = text[i]!;
if (ch === "{" || ch === "[" || ch === "(") depth++;
else if (ch === "}" || ch === "]" || ch === ")") depth--;
else if (depth === 0 && ch === sep) {
parts.push(text.slice(start, i));
start = i + 1;
}
i++;
}
parts.push(text.slice(start));
return parts;
}
/** First index of `ch` at bracket depth 0, skipping `<|"|>` string spans. */
function topLevelIndexOf(text: string, ch: string): number {
let depth = 0;
let i = 0;
const n = text.length;
while (i < n) {
if (text.startsWith(STRING, i)) {
i = skipGemmaString(text, i);
continue;
}
const c = text[i]!;
if (c === "{" || c === "[" || c === "(") depth++;
else if (c === "}" || c === "]" || c === ")") depth--;
else if (depth === 0 && c === ch) return i;
i++;
}
return -1;
}
const grammar: Grammar = {
syntax: "gemma",
prompt: grammarPrompt,
createScanner: () => new GemmaInbandScanner(),
renderToolCall: renderGemmaInvocation,
renderAssistantToolCalls: renderGemmaToolCalls,
renderToolResults: renderGemmaToolResults,
};
export default grammar;
+32
View File
@@ -0,0 +1,32 @@
## Format guide
Emit each call as a `<tool_call>` block. The function name goes on the same line as the opening tag, followed by one `<arg_key>`/`<arg_value>` pair per argument, closed by `</tool_call>`:
```text
<tool_call>get_weather
<arg_key>location</arg_key>
<arg_value>Beijing</arg_value>
<arg_key>days</arg_key>
<arg_value>3</arg_value>
</tool_call>
```
Tool results return in an observation block:
```text
<observation>
<tool_response>
verbatim tool result
</tool_response>
</observation>
```
## Rules
- The name after `<tool_call>` must match a listed function and sit on the same line.
- Emit one `<arg_key>name</arg_key>` + `<arg_value>value</arg_value>` pair per argument; omit unset optional args.
- String values are raw text (no quotes, no escaping); non-string values are valid JSON.
- Multiple calls are consecutive `<tool_call>…</tool_call>` blocks.
- Private reasoning goes in `<think>…</think>`; NEVER put tool calls inside `<think>`.
- Read each `<tool_response>` in call order. NEVER emit `<tool_response>` yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+384
View File
@@ -0,0 +1,384 @@
import {
buildStringArgsResolver,
decodeValue,
mintToolCallId,
partialSuffixOverlap,
partialSuffixOverlapAny,
} from "./coercion";
import grammarPrompt from "./glm.md" with { type: "text" };
import { renderGlmInvocation, renderGlmToolCalls, renderGlmToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
const TOOL_OPEN = "<tool_call>";
const TOOL_CLOSE = "</tool_call>";
const ARG_KEY_OPEN = "<arg_key>";
const ARG_KEY_CLOSE = "</arg_key>";
const ARG_VALUE_OPEN = "<arg_value>";
const ARG_VALUE_CLOSE = "</arg_value>";
const RESPONSE_OPEN = "<tool_response>";
const RESPONSE_CLOSE = "</tool_response>";
const THINK_OPEN = "<think>";
const THINK_CLOSE = "</think>";
const OUTSIDE_TAGS = [
TOOL_OPEN,
ARG_KEY_OPEN,
ARG_KEY_CLOSE,
ARG_VALUE_OPEN,
ARG_VALUE_CLOSE,
RESPONSE_OPEN,
RESPONSE_CLOSE,
THINK_OPEN,
THINK_CLOSE,
] as const;
const OUTSIDE_TAGS_NO_THINK = [
TOOL_OPEN,
ARG_KEY_OPEN,
ARG_KEY_CLOSE,
ARG_VALUE_OPEN,
ARG_VALUE_CLOSE,
RESPONSE_OPEN,
RESPONSE_CLOSE,
] as const;
const BODY_TAGS = [ARG_KEY_OPEN, TOOL_CLOSE] as const;
type State = "outside" | "thinking" | "name" | "body" | "key" | "afterkey" | "value";
interface OpenCall {
id: string;
name: string;
stringArgs: ReadonlySet<string>;
arguments: Record<string, unknown>;
key: string | null;
valueRaw: string;
rawBlock: string;
}
interface TagMatch {
index: number;
tag: string;
}
export class GLMInbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
#call: OpenCall | null = null;
#thinking = "";
#parseThinking: boolean;
#stringArgs: (toolName: string) => ReadonlySet<string>;
constructor(options: InbandScannerOptions = {}) {
this.#parseThinking = options.parseThinking !== false;
this.#stringArgs = options.stringArgs ?? buildStringArgsResolver(options.tools);
}
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
if (!this.#consumeOutside(final, events)) break;
continue;
}
if (this.#state === "thinking") {
this.#consumeThinking(final, events);
if (this.#state === "thinking") break;
continue;
}
if (this.#state === "name") {
if (!this.#consumeName(final, events)) break;
continue;
}
if (this.#state === "body") {
if (!this.#consumeBody(final, events)) break;
continue;
}
if (this.#state === "key") {
if (!this.#consumeKey(final)) break;
continue;
}
if (this.#state === "afterkey") {
if (!this.#consumeAfterKey(final)) break;
continue;
}
if (!this.#consumeValue(final, events)) break;
}
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): boolean {
const tags = this.#parseThinking ? OUTSIDE_TAGS : OUTSIDE_TAGS_NO_THINK;
const match = findFirstTag(this.#buffer, tags);
if (!match) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, tags);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
return false;
}
if (match.index > 0) events.push({ type: "text", text: this.#buffer.slice(0, match.index) });
this.#buffer = this.#buffer.slice(match.index + match.tag.length);
if (match.tag === TOOL_OPEN) {
this.#state = "name";
return true;
}
if (match.tag === THINK_OPEN && this.#parseThinking) {
this.#thinking = "";
events.push({ type: "thinkingStart" });
this.#state = "thinking";
return true;
}
if (match.tag === RESPONSE_OPEN) {
this.#buffer = "";
return false;
}
return true;
}
#consumeThinking(final: boolean, events: InbandScanEvent[]): void {
const close = this.#buffer.indexOf(THINK_CLOSE);
if (close === -1) {
const hold = final ? 0 : partialSuffixOverlap(this.#buffer, THINK_CLOSE);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
this.#emitThinking(emit, events);
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
if (final) this.#endThinking(events);
return;
}
this.#emitThinking(this.#buffer.slice(0, close), events);
this.#buffer = this.#buffer.slice(close + THINK_CLOSE.length);
this.#endThinking(events);
this.#state = "outside";
}
#consumeName(final: boolean, events: InbandScanEvent[]): boolean {
const newline = this.#buffer.indexOf("\n");
const key = this.#buffer.indexOf(ARG_KEY_OPEN);
const close = this.#buffer.indexOf(TOOL_CLOSE);
const delimiter = minFound(newline, key, close);
if (delimiter === -1) {
if (!final) return false;
this.#beginCall(this.#buffer, events);
this.#buffer = "";
this.#endCall(events);
return false;
}
const rawName = this.#buffer.slice(0, delimiter);
this.#beginCall(rawName, events);
if (delimiter === newline) {
this.#appendCallRaw("\n");
this.#buffer = this.#buffer.slice(delimiter + 1);
this.#state = "body";
return true;
}
if (delimiter === key) {
this.#appendCallRaw(ARG_KEY_OPEN);
this.#buffer = this.#buffer.slice(delimiter + ARG_KEY_OPEN.length);
this.#state = "key";
return true;
}
this.#appendCallRaw(TOOL_CLOSE);
this.#buffer = this.#buffer.slice(delimiter + TOOL_CLOSE.length);
this.#endCall(events);
return true;
}
#consumeBody(final: boolean, events: InbandScanEvent[]): boolean {
this.#appendCallRaw(this.#skipWhitespace());
if (this.#buffer.length === 0) return false;
if (this.#buffer.startsWith(ARG_KEY_OPEN)) {
this.#appendCallRaw(ARG_KEY_OPEN);
this.#buffer = this.#buffer.slice(ARG_KEY_OPEN.length);
this.#state = "key";
return true;
}
if (this.#buffer.startsWith(TOOL_CLOSE)) {
this.#appendCallRaw(TOOL_CLOSE);
this.#buffer = this.#buffer.slice(TOOL_CLOSE.length);
this.#endCall(events);
return true;
}
if (!final && partialSuffixOverlapAny(this.#buffer, BODY_TAGS) === this.#buffer.length) return false;
this.#appendCallRaw(this.#buffer[0] ?? "");
this.#buffer = this.#buffer.slice(1);
return true;
}
#consumeKey(final: boolean): boolean {
const close = this.#buffer.indexOf(ARG_KEY_CLOSE);
if (close === -1) {
if (final) this.#dropCall();
return false;
}
if (this.#call) {
this.#call.key = this.#buffer.slice(0, close).trim();
this.#appendCallRaw(this.#buffer.slice(0, close + ARG_KEY_CLOSE.length));
}
this.#buffer = this.#buffer.slice(close + ARG_KEY_CLOSE.length);
this.#state = "afterkey";
return true;
}
#consumeAfterKey(final: boolean): boolean {
this.#appendCallRaw(this.#skipWhitespace());
if (this.#buffer.length === 0) return false;
if (this.#buffer.startsWith(ARG_VALUE_OPEN)) {
this.#appendCallRaw(ARG_VALUE_OPEN);
this.#buffer = this.#buffer.slice(ARG_VALUE_OPEN.length);
if (this.#call) this.#call.valueRaw = "";
this.#state = "value";
return true;
}
if (!final && ARG_VALUE_OPEN.startsWith(this.#buffer)) return false;
this.#appendCallRaw(this.#buffer[0] ?? "");
this.#buffer = this.#buffer.slice(1);
return true;
}
#consumeValue(final: boolean, events: InbandScanEvent[]): boolean {
const close = this.#buffer.indexOf(ARG_VALUE_CLOSE);
if (close === -1) {
const hold = final ? 0 : partialSuffixOverlap(this.#buffer, ARG_VALUE_CLOSE);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
this.#streamValue(emit, events);
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
if (final) this.#dropCall();
return false;
}
this.#streamValue(this.#buffer.slice(0, close), events);
this.#appendCallRaw(ARG_VALUE_CLOSE);
this.#buffer = this.#buffer.slice(close + ARG_VALUE_CLOSE.length);
this.#endValue();
this.#state = "body";
return true;
}
#beginCall(rawName: string, events: InbandScanEvent[]): void {
const name = rawName.trim();
if (name.length === 0) {
this.#dropCall();
return;
}
const id = mintToolCallId();
this.#call = {
id,
name,
stringArgs: this.#stringArgs(name),
arguments: {},
key: null,
valueRaw: "",
rawBlock: `${TOOL_OPEN}${rawName}`,
};
events.push({ type: "toolStart", id, name });
}
#streamValue(chunk: string, events: InbandScanEvent[]): void {
const call = this.#call;
if (!call || call.key === null || chunk.length === 0) return;
call.valueRaw += chunk;
call.rawBlock += chunk;
events.push({ type: "toolArgDelta", id: call.id, name: call.name, key: call.key, delta: chunk });
}
#endValue(): void {
const call = this.#call;
if (!call || call.key === null) return;
call.arguments[call.key] = call.stringArgs.has(call.key) ? call.valueRaw : decodeValue(call.valueRaw);
call.key = null;
call.valueRaw = "";
}
#endCall(events: InbandScanEvent[]): void {
const call = this.#call;
if (!call) {
this.#state = "outside";
return;
}
events.push({
type: "toolEnd",
id: call.id,
name: call.name,
arguments: call.arguments,
rawBlock: call.rawBlock,
});
this.#call = null;
this.#state = "outside";
}
#dropCall(): void {
this.#call = null;
this.#state = "outside";
}
#appendCallRaw(text: string): void {
if (this.#call && text.length > 0) this.#call.rawBlock += text;
}
#emitThinking(delta: string, events: InbandScanEvent[]): void {
if (delta.length === 0) return;
this.#thinking += delta;
events.push({ type: "thinkingDelta", delta });
}
#endThinking(events: InbandScanEvent[]): void {
events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#state = "outside";
}
#skipWhitespace(): string {
let i = 0;
while (i < this.#buffer.length && " \n\t\r".includes(this.#buffer[i]!)) i++;
const skipped = this.#buffer.slice(0, i);
if (i > 0) this.#buffer = this.#buffer.slice(i);
return skipped;
}
}
function findFirstTag(text: string, tags: readonly string[]): TagMatch | null {
let best: TagMatch | null = null;
for (const tag of tags) {
const index = text.indexOf(tag);
if (index === -1) continue;
if (!best || index < best.index) best = { index, tag };
}
return best;
}
function minFound(...values: readonly number[]): number {
let best = -1;
for (const value of values) {
if (value === -1) continue;
if (best === -1 || value < best) best = value;
}
return best;
}
const grammar: Grammar = {
syntax: "glm",
prompt: grammarPrompt,
createScanner: options => new GLMInbandScanner(options),
renderToolCall: renderGlmInvocation,
renderAssistantToolCalls: renderGlmToolCalls,
renderToolResults: renderGlmToolResults,
};
export default grammar;
+30
View File
@@ -0,0 +1,30 @@
## Format guide
Each function call is one assistant message on the `commentary` channel addressed to the function, emitted as text:
```text
<|start|>assistant<|channel|>commentary to=functions.function_name<|message|>{"arg":"value"}<|call|>
```
Put private reasoning in an `analysis` message:
```text
<|start|>assistant<|channel|>analysis<|message|>private reasoning<|end|>
```
Tool results arrive as messages authored by the function, addressed back to the assistant:
```text
<|start|>functions.function_name to=assistant<|channel|>commentary<|message|>verbatim tool result<|end|>
```
## Rules
- Recipient is `functions.` + a listed function name.
- Body is one JSON object matching the schema; omit optional arguments you are not setting.
- Multiple calls = consecutive call messages.
- An optional visible preamble is a `commentary` message ending `<|end|>`.
- NEVER put tool calls in `analysis`.
- NEVER wrap calls in Markdown/code fences.
- Read each tool-result message in call order. NEVER emit tool-result messages yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+272
View File
@@ -0,0 +1,272 @@
import { parseJsonWithRepair } from "../utils/json-parse";
import { asRecord, mintToolCallId, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./harmony.md" with { type: "text" };
import { renderHarmonyInvocation, renderHarmonyToolCalls, renderHarmonyToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner } from "./types";
const START = "<|start|>";
const END = "<|end|>";
const MESSAGE = "<|message|>";
const CHANNEL = "<|channel|>";
const CONSTRAIN = "<|constrain|>";
const RETURN = "<|return|>";
const CALL = "<|call|>";
const ALL_TOKENS = [START, END, MESSAGE, CHANNEL, CONSTRAIN, RETURN, CALL] as const;
const BODY_TOKENS = [END, CALL, RETURN, START, CHANNEL, MESSAGE, CONSTRAIN] as const;
type State = "outside" | "header" | "body";
type BodyMode = "text" | "thinking" | "tool" | "skip";
interface HeaderFields {
role: string;
channel: string;
recipient: string;
}
interface TokenMatch {
index: number;
token: string;
}
export class HarmonyInbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
#mode: BodyMode = "skip";
#id = "";
#name = "";
#toolArgs = "";
#thinking = "";
#rawBlock = "";
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
const next = findNextToken(this.#buffer, ALL_TOKENS);
if (!next) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, ALL_TOKENS);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
break;
}
if (next.index > 0) events.push({ type: "text", text: this.#buffer.slice(0, next.index) });
if (next.token === START) {
this.#rawBlock = START;
this.#buffer = this.#buffer.slice(next.index + START.length);
this.#state = "header";
continue;
}
if (next.token === CHANNEL) {
this.#rawBlock = "";
this.#buffer = this.#buffer.slice(next.index);
this.#state = "header";
continue;
}
this.#buffer = this.#buffer.slice(next.index + next.token.length);
continue;
}
if (this.#state === "header") {
const message = this.#buffer.indexOf(MESSAGE);
if (message === -1) {
if (final) this.#resetAll();
break;
}
const rawHeader = this.#buffer.slice(0, message);
this.#rawBlock += this.#buffer.slice(0, message + MESSAGE.length);
const header = this.#parseHeader(rawHeader);
this.#buffer = this.#buffer.slice(message + MESSAGE.length);
this.#enterBody(header, events);
continue;
}
const next = findNextToken(this.#buffer, BODY_TOKENS);
if (!next) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, BODY_TOKENS);
this.#emitBody(this.#buffer.slice(0, this.#buffer.length - hold), events);
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
if (final && this.#buffer.length === 0) {
this.#finishBody(events);
this.#state = "outside";
}
break;
}
this.#emitBody(this.#buffer.slice(0, next.index), events);
if (next.token === END || next.token === CALL || next.token === RETURN) {
if (this.#mode === "tool") this.#rawBlock += next.token;
this.#buffer = this.#buffer.slice(next.index + next.token.length);
this.#finishBody(events);
this.#state = "outside";
continue;
}
if (next.token === START) {
this.#finishBody(events);
this.#rawBlock = START;
this.#buffer = this.#buffer.slice(next.index + START.length);
this.#state = "header";
continue;
}
if (next.token === CHANNEL) {
this.#finishBody(events);
this.#rawBlock = "";
this.#buffer = this.#buffer.slice(next.index);
this.#state = "header";
continue;
}
if (this.#mode === "tool") this.#rawBlock += next.token;
this.#buffer = this.#buffer.slice(next.index + next.token.length);
}
return events;
}
#enterBody(header: HeaderFields, events: InbandScanEvent[]): void {
this.#clearBody(false);
this.#state = "body";
const assistantMessage = header.role === "" || header.role === "assistant";
if (!assistantMessage) {
this.#mode = "skip";
return;
}
if (header.recipient.length > 0 && header.recipient !== "assistant") {
this.#mode = "tool";
this.#id = mintToolCallId();
this.#name = header.recipient.startsWith("functions.")
? header.recipient.slice("functions.".length)
: header.recipient;
events.push({ type: "toolStart", id: this.#id, name: this.#name });
return;
}
if (header.channel === "analysis") {
this.#mode = "thinking";
events.push({ type: "thinkingStart" });
return;
}
this.#mode = "text";
}
#emitBody(chunk: string, events: InbandScanEvent[]): void {
if (chunk.length === 0) return;
if (this.#mode === "text") {
events.push({ type: "text", text: chunk });
return;
}
if (this.#mode === "thinking") {
this.#thinking += chunk;
events.push({ type: "thinkingDelta", delta: chunk });
return;
}
if (this.#mode === "tool") {
this.#rawBlock += chunk;
this.#toolArgs += chunk;
}
}
#finishBody(events: InbandScanEvent[]): void {
if (this.#mode === "thinking") {
events.push({ type: "thinkingEnd", thinking: this.#thinking });
} else if (this.#mode === "tool" && this.#name.length > 0) {
events.push({
type: "toolEnd",
id: this.#id,
name: this.#name,
arguments: this.#parseArgs(),
rawBlock: this.#rawBlock,
});
}
this.#clearBody();
}
#parseHeader(rawHeader: string): HeaderFields {
const channelIndex = rawHeader.indexOf(CHANNEL);
const rolePart = channelIndex === -1 ? rawHeader : rawHeader.slice(0, channelIndex);
const channelPart = channelIndex === -1 ? "" : rawHeader.slice(channelIndex + CHANNEL.length);
return {
role: firstWord(rolePart),
channel: firstWord(channelPart),
recipient: parseRecipient(rawHeader),
};
}
#parseArgs(): Record<string, unknown> {
const raw = this.#toolArgs.trim();
if (raw.length === 0) return {};
try {
return asRecord(parseJsonWithRepair<unknown>(raw));
} catch {
return {};
}
}
#clearBody(resetRawBlock = true): void {
this.#mode = "skip";
this.#id = "";
this.#name = "";
this.#toolArgs = "";
this.#thinking = "";
if (resetRawBlock) this.#rawBlock = "";
}
#resetAll(): void {
this.#buffer = "";
this.#state = "outside";
this.#clearBody();
}
}
function findNextToken(text: string, tokens: readonly string[]): TokenMatch | undefined {
let match: TokenMatch | undefined;
for (const token of tokens) {
const index = text.indexOf(token);
if (index !== -1 && (!match || index < match.index)) match = { index, token };
}
return match;
}
function firstWord(text: string): string {
const trimmed = text.trimStart();
let end = 0;
while (end < trimmed.length) {
const ch = trimmed[end]!;
if (ch === "<" || /\s/.test(ch)) break;
end++;
}
return trimmed.slice(0, end);
}
function parseRecipient(header: string): string {
const match = /(?:^|\s)to=([^\s<]+)/.exec(header);
return match?.[1] ?? "";
}
const grammar: Grammar = {
syntax: "harmony",
prompt: grammarPrompt,
createScanner: () => new HarmonyInbandScanner(),
renderToolCall: renderHarmonyInvocation,
renderAssistantToolCalls: renderHarmonyToolCalls,
renderToolResults: renderHarmonyToolResults,
};
export default grammar;
+24
View File
@@ -0,0 +1,24 @@
## Format guide
Emit each call as a `<tool_call>` block wrapping a single-line JSON object with `name` and `arguments`:
```text
<tool_call>
{"name":"function_name","arguments":{"arg":"value"}}
</tool_call>
```
Results arrive later as `<tool_response>` blocks:
```text
<tool_response>
verbatim tool result
</tool_response>
```
## Rules
- `name` MUST match a listed function; `arguments` is a JSON object, never a stringified JSON.
- Emit multiple calls as consecutive `<tool_call>` blocks; keep any prose outside them.
- Read each `<tool_response>` in call order. NEVER emit `<tool_response>` yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+171
View File
@@ -0,0 +1,171 @@
import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
import { asRecord, mintToolCallId, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./hermes.md" with { type: "text" };
import { renderHermesInvocation, renderHermesToolCalls, renderToolResponseResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
const TOOL_OPEN = "<tool_call>";
const TOOL_CLOSE = "</tool_call>";
const THINK_OPEN = "<think>";
const THINK_CLOSE = "</think>";
const HOLD_TAGS = [TOOL_OPEN, TOOL_CLOSE, THINK_OPEN, THINK_CLOSE] as const;
export class HermesInbandScanner implements InbandScanner {
#buffer = "";
#inside = false;
#id = "";
#name = "";
#started = false;
#parseThinking: boolean;
#inThinking = false;
#thinking = "";
constructor(options: InbandScannerOptions = {}) {
this.#parseThinking = options.parseThinking === true;
}
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#inThinking) {
const closeThink = this.#buffer.indexOf(THINK_CLOSE);
if (closeThink === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, [THINK_CLOSE]);
const thinking = this.#buffer.slice(0, this.#buffer.length - hold);
if (thinking.length > 0) {
this.#thinking += thinking;
events.push({ type: "thinkingDelta", delta: thinking });
}
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
if (final) {
events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#inThinking = false;
}
break;
}
const thinking = this.#buffer.slice(0, closeThink);
if (thinking.length > 0) {
this.#thinking += thinking;
events.push({ type: "thinkingDelta", delta: thinking });
}
this.#buffer = this.#buffer.slice(closeThink + THINK_CLOSE.length);
events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#inThinking = false;
continue;
}
if (!this.#inside) {
const open = this.#buffer.indexOf(TOOL_OPEN);
const think = this.#parseThinking ? this.#buffer.indexOf(THINK_OPEN) : -1;
const start = open === -1 ? think : think === -1 ? open : Math.min(open, think);
if (start === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, HOLD_TAGS);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
break;
}
if (start > 0) events.push({ type: "text", text: this.#buffer.slice(0, start) });
if (start === think) {
this.#buffer = this.#buffer.slice(start + THINK_OPEN.length);
this.#inThinking = true;
this.#thinking = "";
events.push({ type: "thinkingStart" });
continue;
}
this.#buffer = this.#buffer.slice(start + TOOL_OPEN.length);
this.#inside = true;
this.#id = mintToolCallId();
this.#name = "";
this.#started = false;
continue;
}
const close = this.#buffer.indexOf(TOOL_CLOSE);
const body = close === -1 ? this.#buffer : this.#buffer.slice(0, close);
if (!this.#started) this.#tryStart(body, events);
if (close === -1) {
if (final) this.#reset();
break;
}
const parsed = this.#parseCall(body);
if (parsed) {
if (!this.#started) {
events.push({ type: "toolStart", id: this.#id, name: parsed.name });
this.#started = true;
}
events.push({
type: "toolEnd",
id: this.#id,
name: parsed.name,
arguments: parsed.arguments,
rawBlock: `${TOOL_OPEN}${body}${TOOL_CLOSE}`,
});
}
this.#buffer = this.#buffer.slice(close + TOOL_CLOSE.length);
this.#reset();
}
return events;
}
#tryStart(body: string, events: InbandScanEvent[]): void {
try {
const partial = parseStreamingJson<{ name?: unknown }>(body);
if (typeof partial.name !== "string" || partial.name.length === 0) return;
this.#name = partial.name;
this.#started = true;
events.push({ type: "toolStart", id: this.#id, name: this.#name });
} catch {
// Partial JSON is allowed until the closing tag arrives.
}
}
#parseCall(body: string): { name: string; arguments: Record<string, unknown> } | undefined {
try {
const parsed = parseJsonWithRepair<{ name?: unknown; arguments?: unknown }>(body.trim());
if (typeof parsed.name !== "string" || parsed.name.length === 0) return undefined;
let args = parsed.arguments;
if (typeof args === "string") {
try {
args = parseJsonWithRepair<unknown>(args);
} catch {
args = {};
}
}
return { name: parsed.name, arguments: asRecord(args) };
} catch {
return undefined;
}
}
#reset(): void {
this.#inside = false;
this.#id = "";
this.#name = "";
this.#started = false;
}
}
const grammar: Grammar = {
syntax: "hermes",
prompt: grammarPrompt,
createScanner: options => new HermesInbandScanner(options),
renderToolCall: renderHermesInvocation,
renderAssistantToolCalls: renderHermesToolCalls,
renderToolResults: renderToolResponseResults,
};
export default grammar;
+81
View File
@@ -0,0 +1,81 @@
import type {
AssistantMessage,
Context,
ImageContent,
Message,
TextContent,
ToolCall,
ToolResultMessage,
} from "../types";
import { getInbandGrammar } from "./factory";
import type { Grammar, GrammarToolResult, InbandTool, ToolCallSyntax } from "./types";
export function encodeInbandToolHistory(
messages: Context["messages"],
syntax: ToolCallSyntax,
tools: readonly InbandTool[] = [],
): Context["messages"] {
const grammar = getInbandGrammar(syntax);
const out: Message[] = [];
for (let i = 0; i < messages.length; i++) {
const message = messages[i]!;
if (message.role === "assistant") {
out.push(encodeAssistantMessage(message, grammar, tools));
continue;
}
if (message.role === "toolResult") {
const run: ToolResultMessage[] = [];
let j = i;
while (j < messages.length && messages[j]!.role === "toolResult") {
run.push(messages[j] as ToolResultMessage);
j++;
}
out.push(encodeToolResults(run, grammar));
i = j - 1;
continue;
}
out.push(message);
}
return out;
}
function encodeAssistantMessage(
message: AssistantMessage,
grammar: Grammar,
tools: readonly InbandTool[],
): AssistantMessage {
const toolCalls = message.content.filter((block): block is ToolCall => block.type === "toolCall");
if (toolCalls.length === 0) return message;
const prose = message.content
.filter((block): block is TextContent => block.type === "text")
.map(block => block.text)
.join("\n");
const rendered = grammar.renderAssistantToolCalls(toolCalls, { tools });
const text = prose.trim().length > 0 ? `${prose.trimEnd()}\n${rendered}` : rendered;
return { ...message, content: [{ type: "text", text }] };
}
function encodeToolResults(results: readonly ToolResultMessage[], grammar: Grammar): Message {
const grammarResults: GrammarToolResult[] = [];
const images: ImageContent[] = [];
for (let index = 0; index < results.length; index++) {
const result = results[index]!;
let text = "";
for (const block of result.content) {
if (block.type === "text") text += block.text;
else if (block.type === "image") images.push(block);
}
grammarResults.push({
id: result.toolCallId,
name: result.toolName,
index,
text,
isError: result.isError,
});
}
const content: (TextContent | ImageContent)[] = [
{ type: "text", text: grammar.renderToolResults(grammarResults) },
...images,
];
return { role: "user", content, timestamp: results[0]?.timestamp ?? Date.now() };
}
+8
View File
@@ -0,0 +1,8 @@
export * from "./catalog";
export * from "./coercion";
export * from "./examples";
export * from "./factory";
export * from "./history";
export * from "./inventory";
export * from "./owned-stream";
export * from "./types";
+28
View File
@@ -0,0 +1,28 @@
import { preferredToolSyntax } from "@oh-my-pi/pi-catalog/identity";
import { jsonSchemaToTypeScript, toolWireSchema } from "../utils/schema";
import { renderToolExamples } from "./examples";
import type { InbandTool } from "./types";
/**
* Human-readable per-tool inventory: each tool renders as a `# Tool: <name>`
* section with its description, a simplified TypeScript-style parameter
* signature (derived from the wire JSON Schema), and examples in the model's
* native tool-call syntax. Shared by the verbose system-prompt inventory and
* `/dump` so both render the catalog the same way.
*
* `model` is a model id; the native example syntax is resolved from it
* (`preferredToolSyntax`, which falls back to XML for empty/unknown ids).
*/
export function renderToolInventory(tools: readonly InbandTool[], model: string): string {
if (tools.length === 0) return "";
const syntax = preferredToolSyntax(model);
return tools
.map(tool => {
const params = jsonSchemaToTypeScript(toolWireSchema(tool));
const examples = renderToolExamples(tool, syntax);
const parts = [`# Tool: ${tool.name}`, tool.description ?? "", "", `Parameters: ${params}`];
if (examples) parts.push("", examples);
return parts.join("\n");
})
.join("\n\n");
}
+23
View File
@@ -0,0 +1,23 @@
## Format guide
Emit every call of a turn inside one section. Each call is an id of the fixed form `functions.NAME:INDEX` followed by one JSON arguments object:
```text
<|tool_calls_section_begin|><|tool_call_begin|>functions.NAME:INDEX<|tool_call_argument_begin|>{"arg":"value"}<|tool_call_end|><|tool_calls_section_end|>
```
Tool results arrive later as turns whose body is a `## Return of functions.NAME:INDEX` header then the verbatim result:
```text
<|im_system|>NAME<|im_middle|>## Return of functions.NAME:INDEX
verbatim tool result<|im_end|>
```
## Rules
- `NAME` MUST match a listed function exactly.
- Arguments MUST be one JSON object with double-quoted keys.
- Multiple calls = consecutive `<|tool_call_begin|>…<|tool_call_end|>` blocks in the same section; `INDEX` increments from `0`.
- This format has no thinking channel; NEVER emit `<think>` tags.
- Read each result turn in call order. NEVER emit result turns yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+198
View File
@@ -0,0 +1,198 @@
import { parseJsonWithRepair } from "../utils/json-parse";
import { asRecord, normalizeKimiFunctionName, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./kimi.md" with { type: "text" };
import { renderKimiInvocation, renderKimiToolCalls, renderKimiToolResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner } from "./types";
export const KIMI_SECTION_BEGIN = "<|tool_calls_section_begin|>";
export const KIMI_SECTION_END = "<|tool_calls_section_end|>";
export const KIMI_CALL_BEGIN = "<|tool_call_begin|>";
export const KIMI_CALL_END = "<|tool_call_end|>";
export const KIMI_ARG_BEGIN = "<|tool_call_argument_begin|>";
const TOKENS = [KIMI_SECTION_BEGIN, KIMI_SECTION_END, KIMI_CALL_BEGIN, KIMI_CALL_END, KIMI_ARG_BEGIN] as const;
type State = "outside" | "section" | "header" | "args";
export class KimiInbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
#id = "";
#name = "";
#rawBlock = "";
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
if (!this.#consumeOutside(final, events)) break;
continue;
}
if (this.#state === "section") {
if (!this.#consumeSection(final)) break;
continue;
}
if (this.#state === "header") {
if (!this.#consumeHeader(final, events)) break;
continue;
}
if (!this.#consumeArgs(final, events)) break;
}
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): boolean {
const tokenStart = this.#nextTokenIndex();
if (tokenStart === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, TOKENS);
const emitEnd = this.#buffer.length - hold;
if (emitEnd > 0) events.push({ type: "text", text: this.#buffer.slice(0, emitEnd) });
this.#buffer = this.#buffer.slice(emitEnd);
return false;
}
if (tokenStart > 0) events.push({ type: "text", text: this.#buffer.slice(0, tokenStart) });
this.#buffer = this.#buffer.slice(tokenStart);
const token = this.#tokenAtStart();
if (!token) return false;
this.#buffer = this.#buffer.slice(token.length);
if (token === KIMI_SECTION_BEGIN) this.#state = "section";
else events.push({ type: "text", text: token });
return true;
}
#consumeSection(final: boolean): boolean {
this.#skipWhitespace();
if (this.#buffer.length === 0) return false;
const token = this.#tokenAtStart();
if (token === KIMI_SECTION_END) {
this.#buffer = this.#buffer.slice(KIMI_SECTION_END.length);
this.#state = "outside";
return true;
}
if (token === KIMI_CALL_BEGIN) {
this.#buffer = this.#buffer.slice(KIMI_CALL_BEGIN.length);
this.#state = "header";
return true;
}
if (token) {
this.#buffer = this.#buffer.slice(token.length);
return true;
}
if (!final && partialSuffixOverlapAny(this.#buffer, TOKENS) === this.#buffer.length) return false;
this.#buffer = this.#buffer.slice(1);
return true;
}
#consumeHeader(final: boolean, events: InbandScanEvent[]): boolean {
const sep = this.#buffer.indexOf(KIMI_ARG_BEGIN);
if (sep === -1) {
if (final) this.#dropBufferedCall();
return false;
}
const rawHeader = this.#buffer.slice(0, sep);
this.#id = rawHeader.trim();
this.#name = normalizeKimiFunctionName(this.#id);
this.#rawBlock = `${KIMI_CALL_BEGIN}${rawHeader}${KIMI_ARG_BEGIN}`;
events.push({ type: "toolStart", id: this.#id, name: this.#name });
this.#buffer = this.#buffer.slice(sep + KIMI_ARG_BEGIN.length);
this.#state = "args";
return true;
}
#consumeArgs(final: boolean, events: InbandScanEvent[]): boolean {
const end = this.#buffer.indexOf(KIMI_CALL_END);
if (end === -1) {
if (final) this.#dropBufferedCall();
return false;
}
const rawArgsBlock = this.#buffer.slice(0, end);
const rawArgs = rawArgsBlock.trim();
events.push({
type: "toolEnd",
id: this.#id,
name: this.#name,
arguments: this.#parseArgs(rawArgs),
rawBlock: `${this.#rawBlock}${rawArgsBlock}${KIMI_CALL_END}`,
});
this.#buffer = this.#buffer.slice(end + KIMI_CALL_END.length);
this.#resetCall();
this.#state = "section";
return true;
}
#parseArgs(rawArgs: string): Record<string, unknown> {
if (rawArgs.length === 0) return {};
try {
return asRecord(parseJsonWithRepair<unknown>(rawArgs));
} catch {
return {};
}
}
#nextTokenIndex(): number {
let best = -1;
for (const token of TOKENS) {
const index = this.#buffer.indexOf(token);
if (index !== -1 && (best === -1 || index < best)) best = index;
}
return best;
}
#tokenAtStart(): string | undefined {
for (const token of TOKENS) {
if (this.#buffer.startsWith(token)) return token;
}
return undefined;
}
#skipWhitespace(): void {
let i = 0;
while (i < this.#buffer.length && isWhitespace(this.#buffer.charCodeAt(i))) i++;
if (i > 0) this.#buffer = this.#buffer.slice(i);
}
#dropBufferedCall(): void {
this.#buffer = "";
this.#resetCall();
this.#state = "outside";
}
#resetCall(): void {
this.#id = "";
this.#name = "";
this.#rawBlock = "";
}
}
function isWhitespace(cp: number): boolean {
return cp === 0x20 || cp === 0x09 || cp === 0x0a || cp === 0x0d || cp === 0x0b || cp === 0x0c;
}
const grammar: Grammar = {
syntax: "kimi",
prompt: grammarPrompt,
createScanner: () => new KimiInbandScanner(),
renderToolCall: renderKimiInvocation,
renderAssistantToolCalls: renderKimiToolCalls,
renderToolResults: renderKimiToolResults,
};
export default grammar;
+423
View File
@@ -0,0 +1,423 @@
import type {
AssistantMessage,
AssistantMessageEventStream as AssistantMessageEventStreamType,
TextContent,
ThinkingContent,
ToolCall,
} from "../types";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { buildStringArgsResolver } from "./coercion";
import { createInbandScanner } from "./factory";
import type { InbandScanEvent, InbandScanner, InbandTool, ToolCallSyntax } from "./types";
const RESPONSE_OPEN_TOKENS: Record<ToolCallSyntax, readonly string[]> = {
glm: ["<tool_response>"],
hermes: ["<tool_response>"],
kimi: ["<|im_system|>"],
xml: ["<tool_response>"],
anthropic: ["<function_results>", "<tool_response>"],
deepseek: ["<|tool▁outputs▁begin|>", "<|tool▁output▁begin|>"],
harmony: ["<|start|>functions."],
pi: ["<tool_response>"],
qwen3: ["<tool_response>"],
gemini: ["```tool_outputs"],
gemma: ["<|tool_response>"],
};
function firstTokenIndex(text: string, tokens: readonly string[]): number {
let best = -1;
for (const token of tokens) {
const index = text.indexOf(token);
if (index !== -1 && (best === -1 || index < best)) best = index;
}
return best;
}
type OpenText = { index: number } | undefined;
type OpenThinking = { index: number; text: string } | undefined;
export function parseInbandToolMessage(
message: AssistantMessage,
syntax: ToolCallSyntax,
tools: readonly InbandTool[],
): AssistantMessage {
const projector = new InbandStreamProjector(new AssistantMessageEventStream(), tools, syntax, message, false);
for (const block of message.content) {
if (block.type === "text") projector.text(block.text);
else projector.keep(block);
}
return projector.finish(message, false);
}
export function wrapInbandToolStream(
inner: AssistantMessageEventStreamType,
tools: readonly InbandTool[],
syntax: ToolCallSyntax,
onAbort?: () => void,
abortOnFabrication = true,
): AssistantMessageEventStreamType {
const out = new AssistantMessageEventStream();
void (async () => {
try {
let projector: InbandStreamProjector | undefined;
for await (const event of inner) {
switch (event.type) {
case "start":
projector = new InbandStreamProjector(out, tools, syntax, event.partial, true);
break;
case "thinking_start":
projector?.thinkingStart();
break;
case "thinking_delta":
projector?.thinkingDelta(event.delta);
break;
case "thinking_end":
projector?.thinkingEnd();
break;
case "text_delta":
// `text()` returns true once the model starts fabricating its own
// tool result. In abort mode we cut the turn immediately so the
// provider stops spending tokens on the hallucinated continuation; in
// discard mode we keep draining the stream — the projector is now
// stopped, so `finish` (on `done`) drops everything past the boundary.
if (projector?.text(event.delta) && abortOnFabrication) {
projector.finish(event.partial, true);
onAbort?.();
return;
}
break;
case "toolcall_start": {
// Provider emitted a native structured tool call (e.g. Gemini via
// OpenRouter still returns `functionCall` parts even when owned mode
// sends no `tools`). Forward the native lifecycle live so the UI
// streams it; otherwise the turn loses its only actionable content
// and the loop retries forever on a reasoning-only message. The
// projector ignores nameless "ghost" parts and de-conflicts with the
// in-band channel.
const src = event.partial.content[event.contentIndex];
projector?.nativeToolStart(event.contentIndex, src?.type === "toolCall" ? src.name : "");
break;
}
case "toolcall_delta":
projector?.nativeToolDelta(event.contentIndex, event.delta);
break;
case "toolcall_end":
projector?.nativeToolEnd(event.contentIndex, event.toolCall);
break;
case "done":
projector ??= new InbandStreamProjector(out, tools, syntax, event.message, true);
projector.finish(event.message, true);
return;
case "error":
out.push(event);
return;
}
}
} catch (err) {
out.fail(err);
}
})();
return out;
}
class InbandStreamProjector {
readonly #out: AssistantMessageEventStream;
readonly #scanner: InbandScanner;
readonly #emitEvents: boolean;
readonly #responseOpenTokens: readonly string[];
readonly #responseOverlapLength: number;
#partial: AssistantMessage;
#text: OpenText;
#thinking: OpenThinking;
#toolBlocks = new Map<string, { index: number; block: ToolCall; currentKey?: string; rawValue: string }>();
#fedLen = 0;
#stopped = false;
#responsePending = "";
// Provider-native tool calls forwarded live (e.g. Gemini still returns
// `functionCall` parts under owned mode), keyed by the inner stream's
// `contentIndex`. `#toolChannel` records which channel produced the turn's
// first real call so the other is dropped — no double-dispatch, and no
// guessing from emptiness. Nameless "ghost" parts never lock a channel.
#nativeBlocks = new Map<number, { index: number; block: ToolCall }>();
#toolChannel: "native" | "inband" | undefined;
constructor(
out: AssistantMessageEventStream,
tools: readonly InbandTool[],
syntax: ToolCallSyntax,
seed: AssistantMessage,
emitEvents: boolean,
) {
this.#out = out;
this.#emitEvents = emitEvents;
this.#scanner = createInbandScanner(syntax, {
tools,
stringArgs: buildStringArgsResolver(tools),
parseThinking: true,
});
this.#responseOpenTokens = RESPONSE_OPEN_TOKENS[syntax];
this.#responseOverlapLength = Math.max(0, ...this.#responseOpenTokens.map(token => token.length - 1));
this.#partial = { ...seed, content: [] };
if (emitEvents) this.#out.push({ type: "start", partial: this.#partial });
}
keep(block: AssistantMessage["content"][number]): void {
this.#closeText();
this.#closeThinking();
this.#partial.content.push(block);
}
// Forward a native tool call's lifecycle live. `name` comes from the inner
// stream's partial (set at start for well-behaved providers). Empty `name`
// means a not-yet-identified or "ghost" call — skip until `nativeToolEnd`
// can confirm. Once the in-band channel owns the turn, native calls are
// dropped to avoid double-dispatch.
nativeToolStart(srcIndex: number, name: string): void {
if (this.#stopped || !name || this.#toolChannel === "inband") return;
this.#toolChannel = "native";
this.#closeText();
this.#closeThinking();
const block: ToolCall = { type: "toolCall", id: "", name, arguments: {} };
this.#partial.content.push(block);
const index = this.#partial.content.length - 1;
this.#nativeBlocks.set(srcIndex, { index, block });
if (this.#emitEvents) this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial });
}
nativeToolDelta(srcIndex: number, delta: string): void {
if (this.#stopped) return;
const entry = this.#nativeBlocks.get(srcIndex);
if (!entry) return;
if (this.#emitEvents)
this.#out.push({ type: "toolcall_delta", contentIndex: entry.index, delta, partial: this.#partial });
}
nativeToolEnd(srcIndex: number, toolCall: ToolCall): void {
if (this.#stopped) return;
const entry = this.#nativeBlocks.get(srcIndex);
if (entry) {
Object.assign(entry.block, toolCall);
if (this.#emitEvents)
this.#out.push({
type: "toolcall_end",
contentIndex: entry.index,
toolCall: entry.block,
partial: this.#partial,
});
this.#nativeBlocks.delete(srcIndex);
return;
}
// Never streamed (name was empty at start). Salvage a real call whose name
// only arrived now; drop nameless ghosts and anything the in-band channel
// already claimed.
if (!toolCall.name || this.#toolChannel === "inband") return;
this.#toolChannel = "native";
this.#closeText();
this.#closeThinking();
const block: ToolCall = { ...toolCall };
this.#partial.content.push(block);
const index = this.#partial.content.length - 1;
if (this.#emitEvents) {
this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial });
this.#out.push({ type: "toolcall_end", contentIndex: index, toolCall: block, partial: this.#partial });
}
}
text(delta: string): boolean {
if (this.#stopped) return true;
this.#fedLen += delta.length;
const combined = this.#responsePending + delta;
const responseIndex = firstTokenIndex(combined, this.#responseOpenTokens);
if (responseIndex !== -1) {
this.#responsePending = "";
this.#apply(this.#scanner.feed(combined.slice(0, responseIndex)));
this.#stopped = true;
return true;
}
if (combined.length <= this.#responseOverlapLength) {
this.#responsePending = combined;
return false;
}
const emitLength = combined.length - this.#responseOverlapLength;
this.#responsePending = combined.slice(emitLength);
this.#apply(this.#scanner.feed(combined.slice(0, emitLength)));
return false;
}
thinkingStart(): void {
if (this.#stopped) return;
this.#closeText();
if (this.#thinking) return;
const block: ThinkingContent = { type: "thinking", thinking: "" };
this.#partial.content.push(block);
this.#thinking = { index: this.#partial.content.length - 1, text: "" };
if (this.#emitEvents)
this.#out.push({ type: "thinking_start", contentIndex: this.#thinking.index, partial: this.#partial });
}
thinkingDelta(delta: string): void {
if (this.#stopped) return;
if (!this.#thinking) this.thinkingStart();
const thinking = this.#thinking;
if (!thinking) return;
const block = this.#partial.content[thinking.index] as ThinkingContent;
block.thinking += delta;
thinking.text += delta;
if (this.#emitEvents)
this.#out.push({ type: "thinking_delta", contentIndex: thinking.index, delta, partial: this.#partial });
}
thinkingEnd(): void {
this.#closeThinking();
}
finish(message: AssistantMessage, emitDone: boolean): AssistantMessage {
let fullText = "";
for (const block of message.content) if (block.type === "text") fullText += block.text;
if (!this.#stopped && fullText.length > this.#fedLen) this.text(fullText.slice(this.#fedLen));
if (!this.#stopped && this.#responsePending.length > 0) {
this.#apply(this.#scanner.feed(this.#responsePending));
this.#responsePending = "";
}
this.#apply(this.#scanner.flush());
this.#closeText();
this.#closeThinking();
const hasTools = this.#partial.content.some(block => block.type === "toolCall");
const reason =
hasTools && message.stopReason !== "length" ? "toolUse" : message.stopReason === "length" ? "length" : "stop";
const finalMessage: AssistantMessage = { ...message, content: this.#partial.content, stopReason: reason };
if (emitDone) this.#out.push({ type: "done", reason, message: finalMessage });
return finalMessage;
}
#apply(events: InbandScanEvent[]): void {
for (const event of events) {
switch (event.type) {
case "text":
this.#emitText(event.text);
break;
case "thinkingStart":
this.thinkingStart();
break;
case "thinkingDelta":
this.thinkingDelta(event.delta);
break;
case "thinkingEnd":
this.thinkingEnd();
break;
case "toolStart":
this.#beginTool(event);
break;
case "toolArgDelta":
this.#deltaTool(event);
break;
case "toolEnd":
this.#endTool(event);
break;
}
}
}
#emitText(text: string): void {
if (text.length === 0) return;
this.#closeThinking();
if (!this.#text) {
this.#partial.content.push({ type: "text", text: "" });
this.#text = { index: this.#partial.content.length - 1 };
if (this.#emitEvents)
this.#out.push({ type: "text_start", contentIndex: this.#text.index, partial: this.#partial });
}
const block = this.#partial.content[this.#text.index] as TextContent;
block.text += text;
if (this.#emitEvents)
this.#out.push({ type: "text_delta", contentIndex: this.#text.index, delta: text, partial: this.#partial });
}
#closeText(): void {
if (!this.#text) return;
const block = this.#partial.content[this.#text.index] as TextContent;
if (this.#emitEvents) {
this.#out.push({
type: "text_end",
contentIndex: this.#text.index,
content: block.text,
partial: this.#partial,
});
}
this.#text = undefined;
}
#closeThinking(): void {
if (!this.#thinking) return;
const block = this.#partial.content[this.#thinking.index] as ThinkingContent;
if (this.#emitEvents) {
this.#out.push({
type: "thinking_end",
contentIndex: this.#thinking.index,
content: block.thinking,
partial: this.#partial,
});
}
this.#thinking = undefined;
}
#beginTool(event: Extract<InbandScanEvent, { type: "toolStart" }>): void {
// Native owns the turn → drop the in-band call to avoid double-dispatch.
if (this.#toolChannel === "native") return;
this.#toolChannel = "inband";
this.#closeText();
this.#closeThinking();
if (this.#toolBlocks.has(event.id)) return;
const block: ToolCall = { type: "toolCall", id: event.id, name: event.name, arguments: {} };
this.#partial.content.push(block);
const entry = { index: this.#partial.content.length - 1, block, rawValue: "" };
this.#toolBlocks.set(event.id, entry);
if (this.#emitEvents)
this.#out.push({ type: "toolcall_start", contentIndex: entry.index, partial: this.#partial });
}
#deltaTool(event: Extract<InbandScanEvent, { type: "toolArgDelta" }>): void {
let entry = this.#toolBlocks.get(event.id);
if (!entry) {
this.#beginTool({ type: "toolStart", id: event.id, name: event.name });
entry = this.#toolBlocks.get(event.id);
}
if (!entry) return;
if (entry.currentKey !== event.key) {
entry.currentKey = event.key;
entry.rawValue =
typeof entry.block.arguments[event.key] === "string" ? String(entry.block.arguments[event.key]) : "";
}
entry.rawValue += event.delta;
entry.block.arguments[event.key] = entry.rawValue;
if (this.#emitEvents)
this.#out.push({
type: "toolcall_delta",
contentIndex: entry.index,
delta: event.delta,
partial: this.#partial,
});
}
#endTool(event: Extract<InbandScanEvent, { type: "toolEnd" }>): void {
let entry = this.#toolBlocks.get(event.id);
if (!entry) {
this.#beginTool({ type: "toolStart", id: event.id, name: event.name });
entry = this.#toolBlocks.get(event.id);
}
if (!entry) return;
entry.block.name = event.name;
entry.block.arguments = event.arguments;
if (event.rawBlock !== undefined) entry.block.rawBlock = event.rawBlock;
if (this.#emitEvents)
this.#out.push({
type: "toolcall_end",
contentIndex: entry.index,
toolCall: entry.block,
partial: this.#partial,
});
this.#toolBlocks.delete(event.id);
}
}
+49
View File
@@ -0,0 +1,49 @@
## Format guide
A tool call is a `<call:NAME>…</call:NAME>` block (or self-closing `<call:NAME …/>`) written as plain assistant text; arguments are given as tag attributes, child elements, or a verbatim inline body.
```text
<call:read path="src/a.ts" offset=50/>
```
Objects and arrays use child elements, repeating an element for each array item:
```text
<call:configure>
<object>
<y>4</y>
<list>alpha</list>
<list>beta</list>
</object>
</call:configure>
```
A single string argument can fill the body directly:
```text
<call:edit>
*** Begin Patch
...
*** End Patch
</call:edit>
```
Tool results arrive as response blocks, read in call order:
```text
<tool_response>
verbatim tool result
</tool_response>
```
## Rules
- `NAME` must match a listed function; never wrap calls in JSON or fences.
- Use attributes only for top-level scalars; put objects, arrays, and long strings in child elements.
- Strings are verbatim (no quotes, no entity escaping); numbers, booleans, and null are JSON literals.
- An object opens a child block whose scalar subfields may also be attributes; an array repeats its element once per item.
- The inline body fills the first unset string-typed parameter and may contain any raw text except `</call:NAME>`.
- Emit parallel calls as consecutive blocks. NEVER invent call ids; results are positional.
- This format defines no thinking channel; never emit `<think>`.
- Read each `<tool_response>` in call order. NEVER emit `<tool_response>` yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+585
View File
@@ -0,0 +1,585 @@
import type { ToolArgShape } from "./coercion";
import {
buildArgShapes,
coerceValue,
collectSchemaTypes,
getArrayItemSchema,
getObjectProperties,
isArraySchema,
isObjectSchema,
isStringOnlySchema,
mintToolCallId,
partialSuffixOverlapAny,
} from "./coercion";
import grammarPrompt from "./pi.md" with { type: "text" };
import { renderPiNativeInvocation, renderPiNativeToolCalls, renderToolResponseResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
const CALL_PREFIX = "<call:";
const NAME_START = /[A-Za-z_]/;
const NAME_CHAR = /[A-Za-z0-9_-]/;
const EMPTY_STRING_ARGS: ReadonlySet<string> = new Set<string>();
type ScannerState = "outside" | "body";
type BodyMode = "undecided" | "inline" | "members";
type RawAttribute = { name: string; value: string | true };
type OpenTag = {
name: string;
rawAttrs: RawAttribute[];
selfClosing: boolean;
end: number;
};
type MembersResult = {
ok: boolean;
value: Record<string, unknown>;
next: number;
};
type ValueResult = {
ok: boolean;
value: unknown;
next: number;
};
export class PiNativeInbandScanner implements InbandScanner {
#buffer = "";
#state: ScannerState = "outside";
#bodyMode: BodyMode = "undecided";
#id = "";
#name = "";
#args: Record<string, unknown> = {};
#inlineKey = "";
#inlineValue = "";
#inlineLeading = false;
#rawBlock = "";
readonly #argShapes: Map<string, ToolArgShape>;
readonly #stringArgs: (toolName: string) => ReadonlySet<string>;
constructor(options: InbandScannerOptions = {}) {
this.#argShapes = buildArgShapes(options.tools);
this.#stringArgs =
options.stringArgs ?? (toolName => this.#argShapes.get(toolName)?.stringArgs ?? EMPTY_STRING_ARGS);
}
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
if (!this.#consumeOutside(events, final)) break;
continue;
}
if (this.#bodyMode === "undecided") {
const mode = this.#classifyBody(final);
if (!mode) break;
this.#bodyMode = mode;
if (mode === "inline") {
this.#inlineKey = this.#inlineTargetKey() ?? "input";
this.#inlineValue = "";
this.#inlineLeading = true;
}
}
if (this.#bodyMode === "inline") {
if (!this.#consumeInline(events, final)) break;
continue;
}
if (!this.#consumeMembers(events, final)) break;
}
return events;
}
#consumeOutside(events: InbandScanEvent[], final: boolean): boolean {
const open = this.#buffer.indexOf(CALL_PREFIX);
if (open === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, [CALL_PREFIX]);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
return false;
}
if (open > 0) {
events.push({ type: "text", text: this.#buffer.slice(0, open) });
this.#buffer = this.#buffer.slice(open);
}
const tagEnd = findTagEnd(this.#buffer, 0);
if (tagEnd === -1) {
if (final) this.#buffer = "";
return false;
}
const tag = parseCallOpenTag(this.#buffer);
if (!tag) {
events.push({ type: "text", text: this.#buffer[0] ?? "" });
this.#buffer = this.#buffer.slice(1);
return true;
}
this.#beginCall(tag, events);
this.#buffer = this.#buffer.slice(tag.end);
if (tag.selfClosing) {
events.push({
type: "toolEnd",
id: this.#id,
name: this.#name,
arguments: this.#args,
rawBlock: this.#rawBlock,
});
this.#reset();
return true;
}
this.#state = "body";
this.#bodyMode = "undecided";
return true;
}
#beginCall(tag: OpenTag, events: InbandScanEvent[]): void {
this.#id = mintToolCallId();
this.#name = tag.name;
this.#args = coerceAttributes(tag.rawAttrs, this.#shape()?.properties ?? {});
this.#inlineKey = "";
this.#inlineValue = "";
this.#inlineLeading = false;
this.#rawBlock = this.#buffer.slice(0, tag.end);
events.push({ type: "toolStart", id: this.#id, name: this.#name });
}
#classifyBody(final: boolean): BodyMode | undefined {
const closeTag = this.#closeTag();
const first = skipWhitespace(this.#buffer, 0);
const close = this.#buffer.indexOf(closeTag);
const inlineKey = this.#inlineTargetKey();
if (close !== -1 && first >= close) return inlineKey ? "inline" : "members";
if (first >= this.#buffer.length) return undefined;
const fromFirst = this.#buffer.slice(first);
if (!final && closeTag.startsWith(fromFirst)) return undefined;
if (this.#buffer[first] !== "<") return inlineKey ? "inline" : "members";
if (this.#buffer.startsWith(closeTag, first)) return inlineKey ? "inline" : "members";
if (this.#buffer.startsWith("</", first)) return "members";
const elementName = readElementNamePrefix(this.#buffer, first);
if (elementName === undefined) return final ? (inlineKey ? "inline" : "members") : undefined;
if (elementName.length === 0) return inlineKey ? "inline" : "members";
const shape = this.#shape();
if (!shape) return "members";
return Object.hasOwn(shape.properties, elementName) ? "members" : inlineKey ? "inline" : "members";
}
#consumeInline(events: InbandScanEvent[], final: boolean): boolean {
this.#stripInlineLeadingDelimiter(final);
const closeTag = this.#closeTag();
const close = this.#buffer.indexOf(closeTag);
if (close === -1) {
if (final) {
this.#reset();
this.#buffer = "";
return false;
}
const overlap = partialSuffixOverlapAny(this.#buffer, [closeTag]);
let hold = Math.max(1, overlap);
if (overlap > 0) {
const beforeOverlap = this.#buffer.length - overlap - 1;
if (this.#buffer[beforeOverlap] === "\n") {
hold = Math.max(hold, overlap + 1);
if (this.#buffer[beforeOverlap - 1] === "\r") hold = Math.max(hold, overlap + 2);
}
}
const emitLength = this.#buffer.length - hold;
if (emitLength > 0) {
const delta = this.#buffer.slice(0, emitLength);
this.#rawBlock += delta;
this.#emitInlineDelta(delta, events);
this.#buffer = this.#buffer.slice(emitLength);
}
return false;
}
const rawDelta = this.#buffer.slice(0, close);
this.#rawBlock += rawDelta + closeTag;
let delta = rawDelta;
if (delta.endsWith("\r\n")) delta = delta.slice(0, -2);
else if (delta.endsWith("\n")) delta = delta.slice(0, -1);
this.#emitInlineDelta(delta, events);
this.#args[this.#inlineKey] = this.#inlineValue;
events.push({ type: "toolEnd", id: this.#id, name: this.#name, arguments: this.#args, rawBlock: this.#rawBlock });
this.#buffer = this.#buffer.slice(close + closeTag.length);
this.#reset();
return true;
}
#consumeMembers(events: InbandScanEvent[], final: boolean): boolean {
const closeTag = this.#closeTag();
let searchFrom = 0;
while (true) {
const close = this.#buffer.indexOf(closeTag, searchFrom);
if (close === -1) {
if (final) {
this.#reset();
this.#buffer = "";
}
return false;
}
const body = this.#buffer.slice(0, close);
const parsed = parseMembers(body, 0, undefined, this.#shape()?.properties ?? {});
if (!parsed.ok || skipWhitespace(body, parsed.next) !== body.length) {
searchFrom = close + closeTag.length;
continue;
}
const bodyArgs = parsed.value;
const args = { ...this.#args, ...bodyArgs };
this.#rawBlock += body + closeTag;
this.#emitCompletedStringDeltas(bodyArgs, events);
events.push({ type: "toolEnd", id: this.#id, name: this.#name, arguments: args, rawBlock: this.#rawBlock });
this.#buffer = this.#buffer.slice(close + closeTag.length);
this.#reset();
return true;
}
}
#emitCompletedStringDeltas(args: Record<string, unknown>, events: InbandScanEvent[]): void {
for (const key in args) {
const value = args[key];
if (typeof value === "string" && value.length > 0) {
events.push({ type: "toolArgDelta", id: this.#id, name: this.#name, key, delta: value });
}
}
}
#stripInlineLeadingDelimiter(final: boolean): void {
if (!this.#inlineLeading) return;
if (this.#buffer.length === 0) return;
if (this.#buffer[0] === "\r") {
if (this.#buffer.length === 1 && !final) return;
if (this.#buffer[1] === "\n") {
this.#rawBlock += this.#buffer.slice(0, 2);
this.#buffer = this.#buffer.slice(2);
}
this.#inlineLeading = false;
return;
}
if (this.#buffer[0] === "\n") {
this.#rawBlock += this.#buffer[0];
this.#buffer = this.#buffer.slice(1);
}
this.#inlineLeading = false;
}
#emitInlineDelta(delta: string, events: InbandScanEvent[]): void {
if (delta.length === 0) return;
this.#inlineValue += delta;
events.push({ type: "toolArgDelta", id: this.#id, name: this.#name, key: this.#inlineKey, delta });
}
#inlineTargetKey(): string | undefined {
const shape = this.#shape();
if (shape) {
for (const key of shape.parameterOrder) {
if (Object.hasOwn(this.#args, key)) continue;
return isStringOnlySchema(shape.properties[key]) ? key : undefined;
}
return undefined;
}
for (const key of this.#stringArgs(this.#name)) {
if (!Object.hasOwn(this.#args, key)) return key;
}
return "input";
}
#shape(): ToolArgShape | undefined {
return this.#argShapes.get(this.#name);
}
#closeTag(): string {
return `</call:${this.#name}>`;
}
#reset(): void {
this.#state = "outside";
this.#bodyMode = "undecided";
this.#id = "";
this.#name = "";
this.#args = {};
this.#inlineKey = "";
this.#inlineValue = "";
this.#inlineLeading = false;
this.#rawBlock = "";
}
}
function parseMembers(
text: string,
position: number,
endTag: string | undefined,
properties: Record<string, unknown>,
): MembersResult {
const value: Record<string, unknown> = {};
let index = position;
while (index < text.length) {
index = skipWhitespace(text, index);
if (endTag && text.startsWith(endTag, index)) return { ok: true, value, next: index + endTag.length };
if (index >= text.length) break;
if (text[index] !== "<" || text.startsWith("</", index) || text.startsWith(CALL_PREFIX, index)) {
return { ok: false, value, next: index };
}
const tag = parseElementOpenTag(text, index);
if (!tag) return { ok: false, value, next: index };
const propertySchema = properties[tag.name];
const schemaArray = isArraySchema(propertySchema);
const itemSchema = schemaArray ? getArrayItemSchema(propertySchema) : propertySchema;
const parsed = parseElementValue(text, tag, itemSchema);
if (!parsed.ok) return { ok: false, value, next: index };
addMember(value, tag.name, parsed.value, schemaArray);
index = parsed.next;
}
return endTag ? { ok: false, value, next: index } : { ok: true, value, next: index };
}
function parseElementValue(text: string, tag: OpenTag, schema: unknown): ValueResult {
const attrProperties = getObjectProperties(schema);
const attrs = coerceAttributes(tag.rawAttrs, attrProperties);
if (tag.selfClosing) {
if (isObjectSchema(schema) || tag.rawAttrs.length > 0) return { ok: true, value: attrs, next: tag.end };
return { ok: true, value: coerceValue("", schema), next: tag.end };
}
const bodyStart = tag.end;
const closeTag = `</${tag.name}>`;
if (shouldParseObjectBody(text, bodyStart, closeTag, schema, tag.rawAttrs.length > 0)) {
const parsed = parseMembers(text, bodyStart, closeTag, attrProperties);
if (!parsed.ok) return { ok: false, value: undefined, next: bodyStart };
return { ok: true, value: { ...attrs, ...parsed.value }, next: parsed.next };
}
const close = text.indexOf(closeTag, bodyStart);
if (close === -1) return { ok: false, value: undefined, next: bodyStart };
const raw = stripBlockDelimiters(text.slice(bodyStart, close));
return { ok: true, value: coerceValue(raw, schema), next: close + closeTag.length };
}
function shouldParseObjectBody(
text: string,
bodyStart: number,
closeTag: string,
schema: unknown,
hasAttrs: boolean,
): boolean {
if (isObjectSchema(schema)) return true;
if (isTypedScalarSchema(schema)) return false;
if (hasAttrs) return true;
const first = skipWhitespace(text, bodyStart);
if (text.startsWith(closeTag, first)) return false;
return text[first] === "<" && !text.startsWith("</", first) && !text.startsWith(CALL_PREFIX, first);
}
function isTypedScalarSchema(schema: unknown): boolean {
const types = collectSchemaTypes(schema);
if (types.size === 0) return false;
return !types.has("object") && !types.has("array");
}
function addMember(target: Record<string, unknown>, key: string, value: unknown, schemaArray: boolean): void {
if (schemaArray) {
const existing = target[key];
if (Array.isArray(existing)) existing.push(value);
else target[key] = [value];
return;
}
if (!Object.hasOwn(target, key)) {
target[key] = value;
return;
}
const existing = target[key];
if (Array.isArray(existing)) existing.push(value);
else target[key] = [existing, value];
}
function coerceAttributes(
rawAttrs: readonly RawAttribute[],
properties: Record<string, unknown>,
): Record<string, unknown> {
const attrs: Record<string, unknown> = {};
for (const attr of rawAttrs) {
attrs[attr.name] = attr.value === true ? true : coerceValue(attr.value, properties[attr.name]);
}
return attrs;
}
function parseCallOpenTag(text: string): OpenTag | undefined {
if (!text.startsWith(CALL_PREFIX)) return undefined;
const tagEnd = findTagEnd(text, 0);
if (tagEnd === -1) return undefined;
return parseOpenTagContent(text, CALL_PREFIX.length, tagEnd);
}
function parseElementOpenTag(text: string, start: number): OpenTag | undefined {
if (text[start] !== "<" || text.startsWith("</", start) || text.startsWith(CALL_PREFIX, start)) return undefined;
const tagEnd = findTagEnd(text, start);
if (tagEnd === -1) return undefined;
return parseOpenTagContent(text, start + 1, tagEnd);
}
function parseOpenTagContent(text: string, contentStart: number, tagEnd: number): OpenTag | undefined {
let contentEnd = tagEnd;
let cursor = skipWhitespace(text, contentStart);
const nameStart = cursor;
if (!isNameStart(text[cursor])) return undefined;
cursor++;
while (cursor < contentEnd && isNameChar(text[cursor])) cursor++;
const name = text.slice(nameStart, cursor);
let selfClosing = false;
let last = contentEnd - 1;
while (last >= cursor && isWhitespace(text[last])) last--;
if (text[last] === "/") {
selfClosing = true;
contentEnd = last;
}
return {
name,
rawAttrs: parseRawAttributes(text.slice(cursor, contentEnd)),
selfClosing,
end: tagEnd + 1,
};
}
function parseRawAttributes(text: string): RawAttribute[] {
const attrs: RawAttribute[] = [];
let index = 0;
while (index < text.length) {
index = skipWhitespace(text, index);
if (index >= text.length) break;
if (!isNameStart(text[index])) {
index++;
continue;
}
const nameStart = index;
index++;
while (index < text.length && isNameChar(text[index])) index++;
const name = text.slice(nameStart, index);
index = skipWhitespace(text, index);
if (text[index] !== "=") {
attrs.push({ name, value: true });
continue;
}
index++;
index = skipWhitespace(text, index);
if (index >= text.length) {
attrs.push({ name, value: "" });
break;
}
const quote = text[index];
if (quote === '"' || quote === "'") {
const valueStart = ++index;
while (index < text.length && text[index] !== quote) index++;
attrs.push({ name, value: text.slice(valueStart, index) });
if (index < text.length) index++;
continue;
}
const valueStart = index;
while (index < text.length && !isWhitespace(text[index])) index++;
attrs.push({ name, value: text.slice(valueStart, index) });
}
return attrs;
}
function readElementNamePrefix(text: string, ltIndex: number): string | undefined {
let index = ltIndex + 1;
if (index >= text.length) return undefined;
if (!isNameStart(text[index])) return "";
const start = index;
index++;
while (index < text.length && isNameChar(text[index])) index++;
if (index >= text.length) return undefined;
const next = text[index];
return isWhitespace(next) || next === "/" || next === ">" ? text.slice(start, index) : "";
}
function findTagEnd(text: string, start: number): number {
let quote = "";
for (let index = start; index < text.length; index++) {
const ch = text[index];
if (quote) {
if (ch === quote) quote = "";
continue;
}
if (ch === '"' || ch === "'") {
quote = ch;
continue;
}
if (ch === ">") return index;
}
return -1;
}
function stripBlockDelimiters(raw: string): string {
let start = 0;
let end = raw.length;
if (raw.startsWith("\r\n")) start = 2;
else if (raw.startsWith("\n")) start = 1;
if (end > start) {
if (raw.endsWith("\r\n")) end -= 2;
else if (raw.endsWith("\n")) end -= 1;
}
return raw.slice(start, end);
}
function skipWhitespace(text: string, index: number): number {
while (index < text.length && isWhitespace(text[index])) index++;
return index;
}
function isWhitespace(ch: string | undefined): boolean {
return ch === " " || ch === "\n" || ch === "\r" || ch === "\t" || ch === "\f";
}
function isNameStart(ch: string | undefined): boolean {
return ch !== undefined && NAME_START.test(ch);
}
function isNameChar(ch: string | undefined): boolean {
return ch !== undefined && NAME_CHAR.test(ch);
}
const grammar: Grammar = {
syntax: "pi",
prompt: grammarPrompt,
createScanner: options => new PiNativeInbandScanner(options),
renderToolCall: renderPiNativeInvocation,
renderAssistantToolCalls: renderPiNativeToolCalls,
renderToolResults: renderToolResponseResults,
};
export default grammar;
@@ -0,0 +1,12 @@
# Tools
You may call one or more functions to assist with the user query.
Tool calls are emitted as text using the exact syntax below, not as native provider tool messages.
Available functions are listed inside `<tools></tools>` as one JSON object per line:
<tools>
{{TOOLS}}
</tools>
{{GRAMMAR}}
+27
View File
@@ -0,0 +1,27 @@
## Format guide
Emit each tool call as one `<tool_call>` block wrapping a single-line JSON object with `name` and a nested `arguments` object:
```text
<tool_call>
{"name":"function_name","arguments":{"arg":"value"}}
</tool_call>
```
Do any private reasoning in `<think>...</think>` before your tool calls.
Tool results arrive later in a user turn:
```text
<tool_response>
verbatim tool result
</tool_response>
```
## Rules
- `name` MUST match a listed function; `arguments` is a JSON object, never a JSON string.
- Multiple calls = consecutive `<tool_call>...</tool_call>` blocks; keep prose outside them.
- NEVER put tool calls inside `<think>`.
- Read each `<tool_response>` in call order. NEVER emit `<tool_response>` yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+203
View File
@@ -0,0 +1,203 @@
import { parseJsonWithRepair } from "../utils/json-parse";
import { asRecord, mintToolCallId, partialSuffixOverlapAny } from "./coercion";
import grammarPrompt from "./qwen3.md" with { type: "text" };
import { renderHermesInvocation, renderHermesToolCalls, renderToolResponseResults } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
const TOOL_OPEN = "<tool_call>";
const TOOL_CLOSE = "</tool_call>";
const THINK_OPEN = "<think>";
const THINK_CLOSE = "</think>";
const TOOL_START_TAGS = [TOOL_OPEN] as const;
const START_TAGS = [TOOL_OPEN, THINK_OPEN] as const;
const THINK_CLOSE_TAGS = [THINK_CLOSE] as const;
const COMPLETE_NAME = /^\s*\{\s*"name"\s*:\s*("(?:\\.|[^"\\])*")/;
type State = "outside" | "thinking" | "tool";
export class Qwen3InbandScanner implements InbandScanner {
#buffer = "";
#state: State = "outside";
#id = "";
#name = "";
#started = false;
#thinking = "";
readonly #parseThinking: boolean;
constructor(options: InbandScannerOptions = {}) {
this.#parseThinking = options.parseThinking !== false;
}
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
return this.#consume(true);
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#state === "outside") {
this.#consumeOutside(final, events);
if (this.#state === "outside") break;
continue;
}
if (this.#state === "thinking") {
this.#consumeThinking(final, events);
if (this.#state === "thinking") break;
continue;
}
this.#consumeTool(final, events);
if (this.#state === "tool") break;
}
return events;
}
#consumeOutside(final: boolean, events: InbandScanEvent[]): void {
const tool = this.#buffer.indexOf(TOOL_OPEN);
const think = this.#parseThinking ? this.#buffer.indexOf(THINK_OPEN) : -1;
let start = tool;
let isThink = false;
if (think !== -1 && (start === -1 || think < start)) {
start = think;
isThink = true;
}
if (start === -1) {
const tags = this.#parseThinking ? START_TAGS : TOOL_START_TAGS;
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, tags);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
return;
}
if (start > 0) events.push({ type: "text", text: this.#buffer.slice(0, start) });
if (isThink) {
this.#buffer = this.#buffer.slice(start + THINK_OPEN.length);
this.#state = "thinking";
this.#thinking = "";
events.push({ type: "thinkingStart" });
return;
}
this.#buffer = this.#buffer.slice(start + TOOL_OPEN.length);
this.#state = "tool";
this.#id = mintToolCallId();
this.#name = "";
this.#started = false;
}
#consumeThinking(final: boolean, events: InbandScanEvent[]): void {
const close = this.#buffer.indexOf(THINK_CLOSE);
if (close === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, THINK_CLOSE_TAGS);
const delta = this.#buffer.slice(0, this.#buffer.length - hold);
this.#emitThinkingDelta(delta, events);
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
if (final) this.#endThinking(events);
return;
}
this.#emitThinkingDelta(this.#buffer.slice(0, close), events);
this.#buffer = this.#buffer.slice(close + THINK_CLOSE.length);
this.#endThinking(events);
}
#consumeTool(final: boolean, events: InbandScanEvent[]): void {
const close = this.#buffer.indexOf(TOOL_CLOSE);
const body = close === -1 ? this.#buffer : this.#buffer.slice(0, close);
if (!this.#started) this.#tryStart(body, events);
if (close === -1) {
if (final) this.#resetTool();
return;
}
const parsed = this.#parseCall(body);
if (parsed) {
if (!this.#started) {
events.push({ type: "toolStart", id: this.#id, name: parsed.name });
this.#started = true;
}
events.push({
type: "toolEnd",
id: this.#id,
name: parsed.name,
arguments: parsed.arguments,
rawBlock: `${TOOL_OPEN}${body}${TOOL_CLOSE}`,
});
}
this.#buffer = this.#buffer.slice(close + TOOL_CLOSE.length);
this.#resetTool();
}
#emitThinkingDelta(delta: string, events: InbandScanEvent[]): void {
if (delta.length === 0) return;
this.#thinking += delta;
events.push({ type: "thinkingDelta", delta });
}
#endThinking(events: InbandScanEvent[]): void {
events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#state = "outside";
}
#tryStart(body: string, events: InbandScanEvent[]): void {
const nameMatch = COMPLETE_NAME.exec(body);
if (!nameMatch) return;
let name: unknown;
try {
name = JSON.parse(nameMatch[1]!);
} catch {
return;
}
if (typeof name !== "string" || name.length === 0) return;
this.#name = name;
this.#started = true;
events.push({ type: "toolStart", id: this.#id, name: this.#name });
}
#parseCall(body: string): { name: string; arguments: Record<string, unknown> } | undefined {
try {
const parsed = parseJsonWithRepair<{ name?: unknown; arguments?: unknown }>(body.trim());
if (typeof parsed.name !== "string" || parsed.name.length === 0) return undefined;
let args = parsed.arguments;
if (typeof args === "string") {
try {
args = parseJsonWithRepair<unknown>(args);
} catch {
args = {};
}
}
return { name: parsed.name, arguments: asRecord(args) };
} catch {
return undefined;
}
}
#resetTool(): void {
this.#state = "outside";
this.#id = "";
this.#name = "";
this.#started = false;
}
}
const grammar: Grammar = {
syntax: "qwen3",
prompt: grammarPrompt,
createScanner: options => new Qwen3InbandScanner(options),
renderToolCall: renderHermesInvocation,
renderAssistantToolCalls: renderHermesToolCalls,
renderToolResults: renderToolResponseResults,
};
export default grammar;
+313
View File
@@ -0,0 +1,313 @@
import type { ToolCall } from "../types";
import {
buildArgShapes,
getArrayItemSchema,
getObjectProperties,
isStringOnlySchema,
type ToolArgShape,
} from "./coercion";
import type { GrammarRenderOptions, GrammarToolResult, InbandTool } from "./types";
const DEEPSEEK_TOOL_CALLS_BEGIN = "<|tool▁calls▁begin|>";
const DEEPSEEK_TOOL_CALLS_END = "<|tool▁calls▁end|>";
const DEEPSEEK_TOOL_CALL_BEGIN = "<|tool▁call▁begin|>";
const DEEPSEEK_TOOL_CALL_END = "<|tool▁call▁end|>";
const DEEPSEEK_TOOL_SEPARATOR = "<|tool▁sep|>";
const DEEPSEEK_TOOL_OUTPUT_BEGIN = "<|tool▁output▁begin|>";
const DEEPSEEK_TOOL_OUTPUT_END = "<|tool▁output▁end|>";
export function renderGlmInvocation(call: ToolCall, options: GrammarRenderOptions = {}): string {
return glmInvocation(call, buildArgShapes(options.tools).get(call.name));
}
function glmInvocation(call: ToolCall, shape: ToolArgShape | undefined): string {
let body = `<tool_call>${call.name}`;
for (const key in call.arguments) {
const value = call.arguments[key];
const rendered = shape?.stringArgs.has(key) && typeof value === "string" ? value : stringifyJson(value);
body += `\n<arg_key>${key}</arg_key>\n<arg_value>${rendered}</arg_value>`;
}
return `${body}\n</tool_call>`;
}
export function renderGlmToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
const shapes = buildArgShapes(options.tools);
return calls.map(call => glmInvocation(call, shapes.get(call.name))).join("\n");
}
export function renderGlmToolResults(results: readonly GrammarToolResult[]): string {
return `<observation>\n${renderToolResponseResults(results)}\n</observation>`;
}
export function renderHermesInvocation(call: ToolCall, _options: GrammarRenderOptions = {}): string {
return `<tool_call>\n${stringifyJson({ name: call.name, arguments: call.arguments })}\n</tool_call>`;
}
export function renderHermesToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
return calls.map(call => renderHermesInvocation(call, options)).join("\n");
}
export function renderKimiInvocation(call: ToolCall, _options: GrammarRenderOptions = {}): string {
return kimiInvocation(call, 0);
}
function kimiInvocation(call: ToolCall, index: number): string {
return `<|tool_call_begin|>${kimiCallId(call.name, call.id, index)}<|tool_call_argument_begin|>${stringifyJson(call.arguments)}<|tool_call_end|>`;
}
export function renderKimiToolCalls(calls: readonly ToolCall[]): string {
if (calls.length === 0) return "";
const body = calls.map((call, index) => kimiInvocation(call, index)).join("");
return `<|tool_calls_section_begin|>${body}<|tool_calls_section_end|>`;
}
export function renderKimiToolResults(results: readonly GrammarToolResult[]): string {
return results
.map(
result =>
`<|im_system|>${result.name}<|im_middle|>## Return of ${kimiCallId(result.name, result.id, result.index)}\n${result.text}<|im_end|>`,
)
.join("");
}
export function renderDeepSeekInvocation(call: ToolCall, _options: GrammarRenderOptions = {}): string {
return `${DEEPSEEK_TOOL_CALL_BEGIN}${call.name}${DEEPSEEK_TOOL_SEPARATOR}${stringifyJson(call.arguments)}${DEEPSEEK_TOOL_CALL_END}`;
}
export function renderDeepSeekToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
if (calls.length === 0) return "";
const body = calls.map(call => renderDeepSeekInvocation(call, options)).join("");
return `${DEEPSEEK_TOOL_CALLS_BEGIN}${body}${DEEPSEEK_TOOL_CALLS_END}`;
}
export function renderDeepSeekToolResults(results: readonly GrammarToolResult[]): string {
return results.map(result => `${DEEPSEEK_TOOL_OUTPUT_BEGIN}${result.text}${DEEPSEEK_TOOL_OUTPUT_END}`).join("\n");
}
export function renderHarmonyInvocation(call: ToolCall, options: GrammarRenderOptions = {}): string {
if (options.example) {
return stringifyJson(call.arguments);
}
return `<|start|>assistant<|channel|>commentary to=${harmonyRecipient(call.name)}<|message|>${stringifyJson(call.arguments)}<|call|>`;
}
export function renderHarmonyToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
return calls.map(call => renderHarmonyInvocation(call, options)).join("");
}
export function renderHarmonyToolResults(results: readonly GrammarToolResult[]): string {
return results
.map(
result =>
`<|start|>${harmonyRecipient(result.name)} to=assistant<|channel|>commentary<|message|>${result.text}<|end|>`,
)
.join("");
}
export function renderAnthropicInvocation(call: ToolCall, options: GrammarRenderOptions = {}): string {
return renderXmlInvoke(call, buildArgShapes(options.tools).get(call.name));
}
export function renderAnthropicToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
if (calls.length === 0) return "";
return `<function_calls>\n${renderXmlInvokes(calls, options.tools ?? [])}\n</function_calls>`;
}
export function renderAnthropicToolResults(results: readonly GrammarToolResult[]): string {
const body = results
.map(result => {
const tag = result.isError ? "error" : "result";
const streamTag = result.isError ? "stderr" : "stdout";
return `<${tag}>\n<tool_name>${escapeXmlText(result.name)}</tool_name>\n<${streamTag}>${result.text}</${streamTag}>\n</${tag}>`;
})
.join("\n");
return `<function_results>\n${body}\n</function_results>`;
}
export function renderXmlInvocation(call: ToolCall, options: GrammarRenderOptions = {}): string {
return renderXmlInvoke(call, buildArgShapes(options.tools).get(call.name));
}
export function renderXmlToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
return renderXmlInvokes(calls, options.tools ?? []);
}
export function renderPiNativeInvocation(call: ToolCall, options: GrammarRenderOptions = {}): string {
return piInvocation(call, buildArgShapes(options.tools).get(call.name));
}
function piInvocation(call: ToolCall, shape: ToolArgShape | undefined): string {
let body = `<call:${call.name}>`;
for (const key in call.arguments) {
body += `\n${renderPiNativeElement(key, call.arguments[key], shape?.properties[key])}`;
}
return `${body}\n</call:${call.name}>`;
}
export function renderPiNativeToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
const shapes = buildArgShapes(options.tools);
return calls.map(call => piInvocation(call, shapes.get(call.name))).join("\n");
}
export function renderToolResponseResults(results: readonly GrammarToolResult[]): string {
return results.map(result => `<tool_response>\n${result.text}\n</tool_response>`).join("\n");
}
function renderXmlInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
for (const key in call.arguments) {
const value = call.arguments[key];
const isString = shape?.stringArgs.has(key) === true;
const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
}
return `${body}</invoke>`;
}
function renderXmlInvokes(calls: readonly ToolCall[], tools: readonly InbandTool[]): string {
const shapes = buildArgShapes(tools);
return calls.map(call => renderXmlInvoke(call, shapes.get(call.name))).join("\n");
}
function renderPiNativeElement(key: string, value: unknown, schema: unknown): string {
if (Array.isArray(value)) {
const itemSchema = getArrayItemSchema(schema);
return value.map(item => renderPiNativeElement(key, item, itemSchema)).join("\n");
}
if (value && typeof value === "object") {
const record = value as Record<string, unknown>;
const properties = getObjectProperties(schema);
let body = `<${key}>`;
for (const childKey in record) {
body += `\n${renderPiNativeElement(childKey, record[childKey], properties[childKey])}`;
}
return `${body}\n</${key}>`;
}
return `<${key}>${renderPiNativeScalar(value, schema)}</${key}>`;
}
function renderPiNativeScalar(value: unknown, schema: unknown): string {
if (typeof value === "string") return value;
if (isStringOnlySchema(schema) && value === null) return "";
return stringifyJson(value);
}
function kimiCallId(name: string, id: string, index: number): string {
const trimmed = id.trim();
return trimmed.startsWith("functions.") ? trimmed : `functions.${name}:${index}`;
}
function harmonyRecipient(name: string): string {
return name.startsWith("functions.") ? name : `functions.${name}`;
}
function stringifyJson(value: unknown): string {
return JSON.stringify(value) ?? "null";
}
function escapeXmlAttr(value: string): string {
return value.replaceAll("&", "&amp;").replaceAll('"', "&quot;").replaceAll("<", "&lt;").replaceAll(">", "&gt;");
}
function escapeXmlText(value: string): string {
return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;");
}
// --- Gemini: Pythonic `tool_code` / `default_api` convention ---
const GEMINI_CODE_OPEN = "```tool_code";
const GEMINI_OUTPUT_OPEN = "```tool_outputs";
const GEMINI_FENCE = "```";
export function renderGeminiInvocation(call: ToolCall, options: GrammarRenderOptions = {}): string {
const kwargs = Object.entries(call.arguments)
.map(([key, value]) => `${key}=${pyValue(value)}`)
.join(", ");
return options.example ? `${call.name}(${kwargs})` : `default_api.${call.name}(${kwargs})`;
}
export function renderGeminiToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
// One call renders bare; parallel calls render as a Python list `[a, b]`.
const body =
calls.length === 1
? renderGeminiInvocation(calls[0]!, options)
: `[${calls.map(call => renderGeminiInvocation(call, options)).join(", ")}]`;
// Examples show the bare call; the live wire form fences it as `tool_code`.
return options.example ? body : `${GEMINI_CODE_OPEN}\n${body}\n${GEMINI_FENCE}`;
}
export function renderGeminiToolResults(results: readonly GrammarToolResult[]): string {
return results.map(result => `${GEMINI_OUTPUT_OPEN}\n${result.text}\n${GEMINI_FENCE}`).join("\n");
}
function pyValue(value: unknown): string {
if (value === null || value === undefined) return "None";
if (typeof value === "boolean") return value ? "True" : "False";
if (typeof value === "number") return Number.isFinite(value) ? String(value) : pyString(String(value));
if (typeof value === "string") return pyString(value);
if (Array.isArray(value)) return `[${value.map(pyValue).join(", ")}]`;
if (typeof value === "object") {
const entries = Object.entries(value as Record<string, unknown>);
return `{${entries.map(([key, val]) => `${pyString(key)}: ${pyValue(val)}`).join(", ")}}`;
}
return pyString(String(value));
}
function pyString(value: string): string {
const escaped = value
.replaceAll("\\", "\\\\")
.replaceAll('"', '\\"')
.replaceAll("\n", "\\n")
.replaceAll("\r", "\\r")
.replaceAll("\t", "\\t");
return `"${escaped}"`;
}
// --- Gemma 4: token-delimited `call:NAME{…}` convention ---
const GEMMA_CALL_OPEN = "<|tool_call>";
const GEMMA_CALL_CLOSE = "<tool_call|>";
const GEMMA_RESPONSE_OPEN = "<|tool_response>";
const GEMMA_RESPONSE_CLOSE = "<tool_response|>";
const GEMMA_STRING = '<|"|>';
export function renderGemmaInvocation(call: ToolCall, _options: GrammarRenderOptions = {}): string {
const args = Object.entries(call.arguments)
.map(([key, value]) => `${key}:${gemmaValue(value)}`)
.join(",");
return `${GEMMA_CALL_OPEN}call:${call.name}{${args}}${GEMMA_CALL_CLOSE}`;
}
export function renderGemmaToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string {
return calls.map(call => renderGemmaInvocation(call, options)).join("");
}
export function renderGemmaToolResults(results: readonly GrammarToolResult[]): string {
return results
.map(
result =>
`${GEMMA_RESPONSE_OPEN}response:${result.name}{output:${gemmaValue(parseMaybeJson(result.text))}}${GEMMA_RESPONSE_CLOSE}`,
)
.join("");
}
function gemmaValue(value: unknown): string {
if (value === null || value === undefined) return "null";
if (typeof value === "boolean") return value ? "true" : "false";
if (typeof value === "number") return String(value);
if (typeof value === "string") return `${GEMMA_STRING}${value}${GEMMA_STRING}`;
if (Array.isArray(value)) return `[${value.map(gemmaValue).join(",")}]`;
if (typeof value === "object") {
const entries = Object.entries(value as Record<string, unknown>);
return `{${entries.map(([key, val]) => `${key}:${gemmaValue(val)}`).join(",")}}`;
}
return `${GEMMA_STRING}${String(value)}${GEMMA_STRING}`;
}
function parseMaybeJson(text: string): unknown {
try {
return JSON.parse(text) as unknown;
} catch {
return text;
}
}
+91
View File
@@ -0,0 +1,91 @@
import { partialSuffixOverlapAny } from "./coercion";
import type { InbandScanEvent, InbandScanner } from "./types";
const THINK_OPEN = "<think>";
const THINK_CLOSE = "</think>";
const THINKING_OPEN = "<thinking>";
const THINKING_CLOSE = "</thinking>";
const TAGS = [
{ open: THINK_OPEN, close: THINK_CLOSE },
{ open: THINKING_OPEN, close: THINKING_CLOSE },
] as const;
const OPENS = [THINK_OPEN, THINKING_OPEN] as const;
type Tag = { readonly open: string; readonly close: string };
export class ThinkingInbandScanner implements InbandScanner {
#buffer = "";
#closeTag = "";
#thinking = "";
feed(text: string): InbandScanEvent[] {
if (text.length === 0) return [];
this.#buffer += text;
return this.#consume(false);
}
flush(): InbandScanEvent[] {
const events = this.#consume(true);
if (this.#buffer.length === 0) return events;
if (this.#closeTag) {
this.#emitThinking(this.#buffer, events);
events.push({ type: "thinkingEnd", thinking: this.#thinking });
} else {
events.push({ type: "text", text: this.#buffer });
}
this.#buffer = "";
this.#closeTag = "";
return events;
}
#consume(final: boolean): InbandScanEvent[] {
const events: InbandScanEvent[] = [];
while (this.#buffer.length > 0) {
if (this.#closeTag) {
const close = this.#buffer.indexOf(this.#closeTag);
if (close === -1) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, [this.#closeTag]);
this.#emitThinking(this.#buffer.slice(0, this.#buffer.length - hold), events);
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
break;
}
this.#emitThinking(this.#buffer.slice(0, close), events);
this.#buffer = this.#buffer.slice(close + this.#closeTag.length);
events.push({ type: "thinkingEnd", thinking: this.#thinking });
this.#thinking = "";
this.#closeTag = "";
continue;
}
const tag = findEarliestOpen(this.#buffer);
if (!tag) {
const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OPENS);
const emit = this.#buffer.slice(0, this.#buffer.length - hold);
if (emit.length > 0) events.push({ type: "text", text: emit });
this.#buffer = this.#buffer.slice(this.#buffer.length - hold);
break;
}
if (tag.index > 0) events.push({ type: "text", text: this.#buffer.slice(0, tag.index) });
this.#buffer = this.#buffer.slice(tag.index + tag.open.length);
this.#closeTag = tag.close;
this.#thinking = "";
events.push({ type: "thinkingStart" });
}
return events;
}
#emitThinking(delta: string, events: InbandScanEvent[]): void {
if (delta.length === 0) return;
this.#thinking += delta;
events.push({ type: "thinkingDelta", delta });
}
}
function findEarliestOpen(buffer: string): (Tag & { index: number }) | undefined {
let best: (Tag & { index: number }) | undefined;
for (const tag of TAGS) {
const index = buffer.indexOf(tag.open);
if (index !== -1 && (!best || index < best.index)) best = { ...tag, index };
}
return best;
}
+55
View File
@@ -0,0 +1,55 @@
import type { ToolCallSyntax } from "@oh-my-pi/pi-catalog/identity";
import type { Context, ToolCall } from "../types";
export type { ToolCallSyntax };
export type InbandScanEvent =
| { type: "text"; text: string }
| { type: "thinkingStart" }
| { type: "thinkingDelta"; delta: string }
| { type: "thinkingEnd"; thinking: string }
| { type: "toolStart"; id: string; name: string }
| { type: "toolArgDelta"; id: string; name: string; key: string; delta: string }
| { type: "toolEnd"; id: string; name: string; arguments: Record<string, unknown>; rawBlock?: string };
export interface InbandScanner {
feed(text: string): InbandScanEvent[];
flush(): InbandScanEvent[];
}
export interface GrammarToolResult {
readonly id: string;
readonly name: string;
readonly index: number;
readonly text: string;
readonly isError: boolean;
}
export interface GrammarRenderOptions {
readonly tools?: readonly InbandTool[];
readonly example?: boolean;
}
export interface Grammar {
readonly syntax: ToolCallSyntax;
readonly prompt: string;
createScanner(options?: InbandScannerOptions): InbandScanner;
/** Render a single tool-call invocation — the inner element only, WITHOUT any parallel-call block envelope (e.g. anthropic's `<function_calls>` / kimi's section wrapper). */
renderToolCall(call: ToolCall, options?: GrammarRenderOptions): string;
/** Render a batch of (parallel) tool calls as one complete block, including whatever envelope the syntax wraps multiple calls in. */
renderAssistantToolCalls(calls: readonly ToolCall[], options?: GrammarRenderOptions): string;
renderToolResults(results: readonly GrammarToolResult[], options?: GrammarRenderOptions): string;
}
export interface InbandScannerOptions {
/** string-typed arg names for a tool → read verbatim. Ignored by JSON-carrying syntaxes. */
stringArgs?: (toolName: string) => ReadonlySet<string>;
/** Full tool schemas for schema-driven syntaxes such as GLM XML and pi-native. */
tools?: readonly InbandTool[];
/** XML only: parse pipe-wrapped DeepSeek DSML tags vs plain Anthropic invoke/parameter tags. */
xmlTagset?: "anthropic" | "dsml";
/** Emit thinking markers as thinking events instead of visible text when the syntax defines them. */
parseThinking?: boolean;
}
export type InbandTool = NonNullable<Context["tools"]>[number];
+22
View File
@@ -0,0 +1,22 @@
## Format guide
A call is one `<invoke>` element whose `<parameter>` children carry its arguments:
```text
<invoke name="fn"><parameter name="arg">value</parameter></invoke>
```
Emit consecutive `<invoke>…</invoke>` blocks for multiple calls; you MAY wrap them in `<tool_calls>…</tool_calls>`. Each call's result arrives as a response block:
```text
<tool_response>
verbatim tool result
</tool_response>
```
## Rules
- `name` MUST match a listed function.
- String values are literal text (no JSON quotes or escaping); non-string values are JSON. Add `string="false"` to a parameter only to force JSON parsing of a value the schema treats as a string.
- Read each `<tool_response>` in call order. NEVER emit `<tool_response>` yourself.
- After emitting your tool calls, YOU MUST EMIT THE STOP SEQUENCE AND HALT.
+33
View File
@@ -0,0 +1,33 @@
import { AnthropicInbandScanner } from "./anthropic";
import { DeepSeekInbandScanner } from "./deepseek";
import { renderToolResponseResults, renderXmlInvocation, renderXmlToolCalls } from "./rendering";
import type { Grammar, InbandScanEvent, InbandScanner, InbandScannerOptions } from "./types";
import grammarPrompt from "./xml.md" with { type: "text" };
export class XmlInbandScanner implements InbandScanner {
readonly #inner: InbandScanner;
constructor(options: InbandScannerOptions = {}) {
this.#inner =
options.xmlTagset === "dsml" ? new DeepSeekInbandScanner(options) : new AnthropicInbandScanner(options);
}
feed(text: string): InbandScanEvent[] {
return this.#inner.feed(text);
}
flush(): InbandScanEvent[] {
return this.#inner.flush();
}
}
const grammar: Grammar = {
syntax: "xml",
prompt: grammarPrompt,
createScanner: options => new XmlInbandScanner(options),
renderToolCall: renderXmlInvocation,
renderAssistantToolCalls: renderXmlToolCalls,
renderToolResults: renderToolResponseResults,
};
export default grammar;
+54 -11
View File
@@ -612,9 +612,9 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
return stream;
};
type ToolCallState = ToolCall & { index: number; partialJson?: string; kind: "mcp" | "todo" };
export type ToolCallState = ToolCall & { index: number; partialJson?: string; kind: "mcp" | "todo" };
interface BlockState {
export interface BlockState {
currentTextBlock: (TextContent & { index: number }) | null;
currentThinkingBlock: (ThinkingContent & { index: number }) | null;
currentToolCall: ToolCallState | null;
@@ -625,7 +625,7 @@ interface BlockState {
setFirstTokenTime: () => void;
}
interface UsageState {
export interface UsageState {
sawTokenDelta: boolean;
}
@@ -1940,7 +1940,40 @@ function buildMcpErrorResult(error: string) {
});
}
function processInteractionUpdate(
/**
* Merge the decoded completion-frame `McpArgs` map into the args assembled
* from streamed `args_text_delta` snapshots.
*
* The completion frame is authoritative for the scalars it carries — but it
* can omit oversized parameters entirely and can downgrade a structured value
* to its raw string fallback when `decodeMcpArgValue` cannot parse it as
* JSON. Overwriting the streamed args wholesale therefore loses data (e.g.
* the task tool's `tasks` array on multi-subagent dispatches, issue #2615).
*
* Rules per key:
* - completion key absent → keep the streamed value.
* - completion is a string while the streamed value is structured (object or
* array) → keep the streamed value (the completion frame downgraded it).
* - otherwise → completion wins.
*/
export function mergeCursorMcpToolCallArgs(
streamed: Record<string, unknown> | undefined,
completion: Record<string, unknown> | undefined,
): Record<string, unknown> {
const merged: Record<string, unknown> = { ...(streamed ?? {}) };
if (!completion) return merged;
for (const [key, completionValue] of Object.entries(completion)) {
const streamedValue = merged[key];
if (typeof completionValue === "string" && streamedValue !== null && typeof streamedValue === "object") {
continue;
}
merged[key] = completionValue;
}
return merged;
}
/** Exported for tests: drives one Cursor interaction update through the streaming state machine. */
export function processInteractionUpdate(
update: any,
output: AssistantMessage,
stream: AssistantMessageEventStream,
@@ -2034,20 +2067,30 @@ function processInteractionUpdate(
}
} else if (updateCase === "toolCallDelta" || updateCase === "partialToolCall") {
if (state.currentToolCall?.kind === "mcp") {
const delta = update.message.value.argsTextDelta || "";
state.currentToolCall.partialJson = `${state.currentToolCall.partialJson ?? ""}${delta}`;
state.currentToolCall.arguments = parseStreamingJson(state.currentToolCall.partialJson ?? "");
// Cursor's `args_text_delta` is "aggregated args text so far" per agent.proto: each
// delta is a cumulative snapshot of the JSON-text args. Strip the prefix we already
// have to recover the new suffix; fall back to treating the value as an incremental
// fragment when it doesn't extend the buffer.
const snapshot: string = update.message.value.argsTextDelta || "";
const current = state.currentToolCall.partialJson ?? "";
const chunk = snapshot.startsWith(current) ? snapshot.slice(current.length) : snapshot;
if (chunk.length === 0) {
return;
}
state.currentToolCall.partialJson = current + chunk;
state.currentToolCall.arguments = parseStreamingJson(state.currentToolCall.partialJson);
const idx = output.content.indexOf(state.currentToolCall);
stream.push({ type: "toolcall_delta", contentIndex: idx, delta, partial: output });
stream.push({ type: "toolcall_delta", contentIndex: idx, delta: chunk, partial: output });
}
} else if (updateCase === "toolCallCompleted") {
if (state.currentToolCall) {
const toolCall = update.message.value.toolCall;
if (state.currentToolCall.kind === "mcp") {
const decodedArgs = decodeMcpArgsMap(toolCall?.mcpToolCall?.args?.args);
if (decodedArgs) {
state.currentToolCall.arguments = decodedArgs;
}
state.currentToolCall.arguments = mergeCursorMcpToolCallArgs(
state.currentToolCall.arguments as Record<string, unknown> | undefined,
decodedArgs,
);
} else if (state.currentToolCall.kind === "todo" && toolCall) {
const todoArgs = buildTodoArgs(toolCall);
if (todoArgs) {
@@ -277,11 +277,39 @@ interface CodexRequestSetup {
websocketFirstEventTimeoutMs: number | undefined;
}
interface CodexOpenItem {
item: CodexEventItem;
block: CodexOutputBlock | null;
/** Index of {@link block} in `output.content`; `-1` when no block was created for this item. */
contentIndex: number;
itemId?: string;
outputIndex?: number;
}
interface CodexStreamRuntime {
eventStream: AsyncGenerator<Record<string, unknown>>;
requestBodyForState: RequestBody;
transport: CodexTransport;
websocketState?: CodexWebSocketSessionState;
/**
* Items open on the wire keyed by `item.id`. `response.output_item.added`
* registers here; `output_item.done` removes. A keyed event whose `item_id`
* is not present is dropped rather than appended to a sibling.
*/
openItems: Map<string, CodexOpenItem>;
/**
* Items open on the wire keyed by `output_index` for streams whose function
* call items omit `id`; these still carry `output_index` on deltas/done.
*/
openItemsByOutputIndex: Map<number, CodexOpenItem>;
/**
* Most recently added open item for events that omit both `item_id` and
* `output_index`. Always tracks the latest `output_item.added`, including
* fully keyless items that never make it into the keyed maps; cleared when
* its item closes.
*/
currentEntry: CodexOpenItem | null;
/** Convenience mirrors of {@link currentEntry} for legacy singleton handlers. */
currentItem: CodexEventItem | null;
currentBlock: CodexOutputBlock | null;
nativeOutputItems: Array<Record<string, unknown>>;
@@ -1076,6 +1104,9 @@ function createCodexStreamRuntime(initial: {
requestBodyForState: initial.requestBodyForState,
transport: initial.transport,
websocketState: initial.websocketState,
openItems: new Map(),
openItemsByOutputIndex: new Map(),
currentEntry: null,
currentItem: null,
currentBlock: null,
nativeOutputItems: [],
@@ -1088,6 +1119,48 @@ function createCodexStreamRuntime(initial: {
};
}
/**
* Wipe per-attempt accumulator state before a recovery path replays the turn.
* Keeps {@link CodexStreamRuntime.openItems} and the legacy singleton-current
* pointers in lockstep with {@link CodexStreamRuntime.nativeOutputItems} so a
* stale delta from the failed attempt can't bind to a sibling on the retry.
*/
function resetCodexStreamAccumulators(runtime: CodexStreamRuntime): void {
runtime.openItems.clear();
runtime.openItemsByOutputIndex.clear();
runtime.currentEntry = null;
runtime.currentItem = null;
runtime.currentBlock = null;
runtime.nativeOutputItems.length = 0;
}
/**
* Look up the open item a Codex stream event targets. `item_id` wins because it
* uniquely identifies a response item; `output_index` covers idless function
* call items. A keyed event whose target is already closed is dropped instead
* of being routed to a sibling. Only streams that omit both keys fall back to
* {@link CodexStreamRuntime.currentEntry} — the most recently added item,
* including fully keyless ones that never reached the keyed maps.
*/
function openItemForEvent(runtime: CodexStreamRuntime, rawEvent: Record<string, unknown>): CodexOpenItem | null {
const itemId = typeof rawEvent.item_id === "string" ? rawEvent.item_id : "";
if (itemId) return runtime.openItems.get(itemId) ?? null;
const outputIndex = readOptionalInteger(rawEvent.output_index);
if (outputIndex !== undefined) return runtime.openItemsByOutputIndex.get(outputIndex) ?? null;
return runtime.currentEntry;
}
function closeCodexOpenItem(runtime: CodexStreamRuntime, entry: CodexOpenItem | null | undefined): void {
if (!entry) return;
if (entry.itemId) runtime.openItems.delete(entry.itemId);
if (entry.outputIndex !== undefined) runtime.openItemsByOutputIndex.delete(entry.outputIndex);
if (runtime.currentEntry === entry) {
runtime.currentEntry = null;
runtime.currentItem = null;
runtime.currentBlock = null;
}
}
function resetWhitespaceToolCallArgumentsDelta(runtime: CodexStreamRuntime): void {
runtime.whitespaceToolCallArgumentsDelta = undefined;
}
@@ -1209,11 +1282,25 @@ function handleCodexStreamEvent(
const item = rawEvent.item as CodexEventItem;
runtime.currentItem = item;
runtime.currentBlock = createOutputBlockForItem(item);
let contentIndex = -1;
if (runtime.currentBlock) {
output.content.push(runtime.currentBlock);
contentIndex = output.content.length - 1;
}
// Track every open item by every stable key the wire gives us. `item.id`
// is best; `output_index` preserves idless function/custom tool calls and
// keeps their final args authoritative when only `output_item.done`
// carries the full payload.
const itemId = typeof (item as { id?: string }).id === "string" ? (item as { id: string }).id : undefined;
const outputIndex = readOptionalInteger(rawEvent.output_index);
const entry: CodexOpenItem = { item, block: runtime.currentBlock, contentIndex, itemId, outputIndex };
runtime.currentEntry = entry;
if (itemId) runtime.openItems.set(itemId, entry);
if (outputIndex !== undefined) runtime.openItemsByOutputIndex.set(outputIndex, entry);
if (!runtime.currentBlock) return firstTokenTime;
output.content.push(runtime.currentBlock);
stream.push({
type: getOutputBlockStartEventType(runtime.currentBlock),
contentIndex: output.content.length - 1,
contentIndex,
partial: output,
});
return firstTokenTime;
@@ -1257,7 +1344,7 @@ function handleCodexStreamEvent(
if (eventType === "response.function_call_arguments.done") {
resetWhitespaceToolCallArgumentsDelta(runtime);
handleToolCallArgumentsDone(runtime.currentItem, runtime.currentBlock, rawEvent);
handleToolCallArgumentsDone(runtime, rawEvent);
return firstTokenTime;
}
@@ -1269,7 +1356,7 @@ function handleCodexStreamEvent(
if (eventType === "response.custom_tool_call_input.done") {
resetWhitespaceToolCallArgumentsDelta(runtime);
handleCustomToolCallInputDone(runtime.currentItem, runtime.currentBlock, rawEvent);
handleCustomToolCallInputDone(runtime, rawEvent);
return firstTokenTime;
}
@@ -1431,36 +1518,37 @@ function handleToolCallArgumentsDelta(
): CodexWhitespaceToolCallArgumentsDeltaInterruption | undefined {
const delta = (rawEvent as { delta?: string }).delta || "";
// Observe BEFORE the item/block guard: degenerate whitespace frames can keep
// arriving after the item closed (currentBlock detached) and still count as
// arriving after the item closed (entry detached) and still count as
// progress for the idle watchdogs — dropping them unobserved would reopen
// the infinite-loop hole the breaker exists for.
const interruption = observeWhitespaceToolCallArgumentsDelta(runtime, rawEvent, delta);
if (interruption) return interruption;
const currentItem = runtime.currentItem;
const currentBlock = runtime.currentBlock;
if (currentItem?.type !== "function_call" || currentBlock?.type !== "toolCall") return undefined;
currentBlock.partialJson += delta;
const throttled = parseStreamingJsonThrottled(currentBlock.partialJson, currentBlock.lastParseLen ?? 0);
// Route to the entry the event keys to; a delta whose item already closed
// is dropped instead of leaking into a sibling tool call (#2619).
const entry = openItemForEvent(runtime, rawEvent);
if (!entry) return undefined;
if (entry.item.type !== "function_call" || entry.block?.type !== "toolCall") return undefined;
const block = entry.block;
block.partialJson += delta;
const throttled = parseStreamingJsonThrottled(block.partialJson, block.lastParseLen ?? 0);
if (throttled) {
currentBlock.arguments = throttled.value;
currentBlock.lastParseLen = throttled.parsedLen;
block.arguments = throttled.value;
block.lastParseLen = throttled.parsedLen;
}
stream.push({ type: "toolcall_delta", contentIndex: output.content.length - 1, delta, partial: output });
stream.push({ type: "toolcall_delta", contentIndex: entry.contentIndex, delta, partial: output });
return undefined;
}
function handleToolCallArgumentsDone(
currentItem: CodexEventItem | null,
currentBlock: CodexOutputBlock | null,
rawEvent: Record<string, unknown>,
): void {
if (currentItem?.type !== "function_call" || currentBlock?.type !== "toolCall") return;
function handleToolCallArgumentsDone(runtime: CodexStreamRuntime, rawEvent: Record<string, unknown>): void {
const entry = openItemForEvent(runtime, rawEvent);
if (entry?.item.type !== "function_call" || entry.block?.type !== "toolCall") return;
const args = (rawEvent as { arguments?: string }).arguments;
if (typeof args === "string") {
currentBlock.partialJson = args;
currentBlock.arguments = parseStreamingJson(currentBlock.partialJson);
delete (currentBlock as { partialJson?: string }).partialJson;
delete (currentBlock as { lastParseLen?: number }).lastParseLen;
const block = entry.block;
block.partialJson = args;
block.arguments = parseStreamingJson(block.partialJson);
delete (block as { partialJson?: string }).partialJson;
delete (block as { lastParseLen?: number }).lastParseLen;
}
}
@@ -1474,25 +1562,23 @@ function handleCustomToolCallInputDelta(
// Observe BEFORE the item/block guard — see handleToolCallArgumentsDelta.
const interruption = observeWhitespaceToolCallArgumentsDelta(runtime, rawEvent, delta);
if (interruption) return interruption;
const currentItem = runtime.currentItem;
const currentBlock = runtime.currentBlock;
if (currentItem?.type !== "custom_tool_call" || currentBlock?.type !== "toolCall") return undefined;
currentBlock.partialJson += delta;
(currentBlock.arguments as { input?: string }).input = currentBlock.partialJson;
stream.push({ type: "toolcall_delta", contentIndex: output.content.length - 1, delta, partial: output });
const entry = openItemForEvent(runtime, rawEvent);
if (!entry) return undefined;
if (entry.item.type !== "custom_tool_call" || entry.block?.type !== "toolCall") return undefined;
const block = entry.block;
block.partialJson += delta;
(block.arguments as { input?: string }).input = block.partialJson;
stream.push({ type: "toolcall_delta", contentIndex: entry.contentIndex, delta, partial: output });
return undefined;
}
function handleCustomToolCallInputDone(
currentItem: CodexEventItem | null,
currentBlock: CodexOutputBlock | null,
rawEvent: Record<string, unknown>,
): void {
if (currentItem?.type !== "custom_tool_call" || currentBlock?.type !== "toolCall") return;
function handleCustomToolCallInputDone(runtime: CodexStreamRuntime, rawEvent: Record<string, unknown>): void {
const entry = openItemForEvent(runtime, rawEvent);
if (entry?.item.type !== "custom_tool_call" || entry.block?.type !== "toolCall") return;
const input = (rawEvent as { input?: string }).input;
if (typeof input === "string") {
currentBlock.partialJson = input;
currentBlock.arguments = { input };
entry.block.partialJson = input;
entry.block.arguments = { input };
}
}
@@ -1508,32 +1594,42 @@ function handleOutputItemDone(
const item = structuredCloneJSON(rawItem) as CodexEventItem;
runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
if (item.type === "reasoning" && runtime.currentBlock?.type === "thinking") {
runtime.currentBlock.thinking = item.summary?.map(summary => summary.text).join("\n\n") || "";
runtime.currentBlock.thinkingSignature = JSON.stringify(item);
// Match the finalization to the OPEN ITEM that started this block, not the
// singleton current — interleaved items can finish out of order, so the
// most-recently-added block may belong to a sibling (#2619). Some Codex
// function/custom tool items omit `id`; in that case `output_index` still
// routes `output_item.done` to the block that received `output_item.added`.
const itemId = typeof (item as { id?: string }).id === "string" ? (item as { id: string }).id : "";
const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? openItemForEvent(runtime, rawEvent);
const block = entry?.block ?? null;
const contentIndex = entry?.contentIndex ?? output.content.length - 1;
if (item.type === "reasoning" && block?.type === "thinking") {
block.thinking = item.summary?.map(summary => summary.text).join("\n\n") || "";
block.thinkingSignature = JSON.stringify(item);
stream.push({
type: "thinking_end",
contentIndex: output.content.length - 1,
content: runtime.currentBlock.thinking,
contentIndex,
content: block.thinking,
partial: output,
});
runtime.currentBlock = null;
closeCodexOpenItem(runtime, entry);
return;
}
if (item.type === "message" && runtime.currentBlock?.type === "text") {
runtime.currentBlock.text = item.content
if (item.type === "message" && block?.type === "text") {
block.text = item.content
.map(content => (content.type === "output_text" ? content.text : content.refusal))
.join("");
const phase = item.phase === "commentary" || item.phase === "final_answer" ? item.phase : undefined;
runtime.currentBlock.textSignature = encodeTextSignatureV1(item.id, phase);
block.textSignature = encodeTextSignatureV1(item.id, phase);
stream.push({
type: "text_end",
contentIndex: output.content.length - 1,
content: runtime.currentBlock.text,
contentIndex,
content: block.text,
partial: output,
});
runtime.currentBlock = null;
closeCodexOpenItem(runtime, entry);
return;
}
@@ -1544,26 +1640,25 @@ function handleOutputItemDone(
name: item.name,
arguments: parseStreamingJson(item.arguments || "{}"),
};
if (runtime.currentBlock?.type === "toolCall") {
if (block?.type === "toolCall") {
// Persist the authoritative final args on the stored block; the throttled
// delta parser may have left currentBlock.arguments stale (often `{}`).
runtime.currentBlock.arguments = toolCall.arguments;
delete (runtime.currentBlock as { partialJson?: string }).partialJson;
delete (runtime.currentBlock as { lastParseLen?: number }).lastParseLen;
// Detach so a late/duplicate arguments.delta cannot append to the
// finished block or trip the whitespace-loop guard against it.
runtime.currentBlock = null;
// delta parser may have left block.arguments stale (often `{}`).
block.arguments = toolCall.arguments;
delete (block as { partialJson?: string }).partialJson;
delete (block as { lastParseLen?: number }).lastParseLen;
}
// Detach so a late/duplicate arguments.delta cannot append to the
// finished block or trip the whitespace-loop guard against it.
closeCodexOpenItem(runtime, entry);
runtime.canSafelyReplayWebsocketOverSse = false;
stream.push({ type: "toolcall_end", contentIndex: output.content.length - 1, toolCall, partial: output });
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
return;
}
if (item.type === "custom_tool_call") {
const rawInput =
runtime.currentBlock?.type === "toolCall" && runtime.currentBlock.partialJson
? runtime.currentBlock.partialJson
: (item.input ?? "");
const partial =
block?.type === "toolCall" ? (block as ToolCall & { partialJson?: string }).partialJson : undefined;
const rawInput = partial && partial.length > 0 ? partial : (item.input ?? "");
const toolCall: ToolCall = {
type: "toolCall",
id: encodeResponsesToolCallId(item.call_id, item.id),
@@ -1571,13 +1666,13 @@ function handleOutputItemDone(
arguments: { input: rawInput },
customWireName: item.name,
};
if (runtime.currentBlock?.type === "toolCall") {
runtime.currentBlock.arguments = { input: rawInput };
delete (runtime.currentBlock as { partialJson?: string }).partialJson;
runtime.currentBlock = null;
if (block?.type === "toolCall") {
block.arguments = { input: rawInput };
delete (block as { partialJson?: string }).partialJson;
}
closeCodexOpenItem(runtime, entry);
runtime.canSafelyReplayWebsocketOverSse = false;
stream.push({ type: "toolcall_end", contentIndex: output.content.length - 1, toolCall, partial: output });
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
return;
}
@@ -1710,8 +1805,7 @@ function dropTrailingDegenerateToolCall(output: AssistantMessage, runtime: Codex
if (block && block.type === "toolCall" && output.content[output.content.length - 1] === block) {
output.content.pop();
}
runtime.currentItem = null;
runtime.currentBlock = null;
closeCodexOpenItem(runtime, runtime.currentEntry);
}
/**
@@ -1757,10 +1851,8 @@ async function tryRecoverCodexWhitespaceToolCallLoop(
transport: runtime.transport,
});
runtime.currentItem = null;
runtime.currentBlock = null;
resetCodexStreamAccumulators(runtime);
runtime.sawTerminalEvent = false;
runtime.nativeOutputItems.length = 0;
resetWhitespaceToolCallArgumentsDelta(runtime);
resetOutputState(context.output);
context.firstTokenTime = undefined;
@@ -1817,9 +1909,7 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
if (context.output.content.length > 0) {
// Content already emitted to the caller — cannot safely continue on a new WS.
// Reset and replay the full request over SSE.
runtime.currentItem = null;
runtime.currentBlock = null;
runtime.nativeOutputItems.length = 0;
resetCodexStreamAccumulators(runtime);
resetOutputState(context.output);
context.firstTokenTime = undefined;
recordCodexWebSocketFailure(websocketState, true);
@@ -1832,9 +1922,7 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
// over websocket, bounded by the shared retry budget: an account-scoped
// limit can reject every fresh connection, and an unbounded loop would
// hammer the endpoint with zero backoff.
runtime.currentItem = null;
runtime.currentBlock = null;
runtime.nativeOutputItems.length = 0;
resetCodexStreamAccumulators(runtime);
context.firstTokenTime = undefined;
if (runtime.websocketStreamRetries >= getCodexWebSocketRetryBudget()) {
recordCodexWebSocketFailure(websocketState, true);
@@ -1886,10 +1974,8 @@ async function tryRecoverCodexPreviousResponseNotFound(
runtime.providerRetryAttempt += 1;
resetCodexWebSocketAppendState(websocketState);
resetCodexSessionMetadata(websocketState);
runtime.currentItem = null;
runtime.currentBlock = null;
resetCodexStreamAccumulators(runtime);
runtime.sawTerminalEvent = false;
runtime.nativeOutputItems.length = 0;
resetOutputState(context.output);
context.firstTokenTime = undefined;
@@ -1936,9 +2022,7 @@ async function tryReplayWebsocketFailureOverSse(
// Full re-send on a fresh socket: clear accumulator state from the failed
// attempt. Content is empty here, but blockless native items (e.g.
// web_search_call) may already have accumulated.
runtime.currentItem = null;
runtime.currentBlock = null;
runtime.nativeOutputItems.length = 0;
resetCodexStreamAccumulators(runtime);
context.firstTokenTime = undefined;
await scheduler.wait(getCodexWebSocketRetryDelayMs(runtime.websocketStreamRetries), {
signal: context.requestSetup.requestSignal,
@@ -1947,9 +2031,7 @@ async function tryReplayWebsocketFailureOverSse(
return true;
}
runtime.currentItem = null;
runtime.currentBlock = null;
runtime.nativeOutputItems.length = 0;
resetCodexStreamAccumulators(runtime);
resetOutputState(context.output);
context.firstTokenTime = undefined;
@@ -1985,10 +2067,8 @@ async function tryRetryCodexProviderError(
transport: runtime.transport,
});
runtime.currentItem = null;
runtime.currentBlock = null;
resetCodexStreamAccumulators(runtime);
runtime.sawTerminalEvent = false;
runtime.nativeOutputItems.length = 0;
resetOutputState(context.output);
context.firstTokenTime = undefined;
await scheduler.wait(CODEX_RETRY_DELAY_MS * runtime.providerRetryAttempt, {
+6
View File
@@ -0,0 +1,6 @@
import type { ProviderDefinition } from "./types";
export const azureProvider = {
id: "azure",
name: "Azure OpenAI",
} as const satisfies ProviderDefinition;
+2
View File
@@ -3,6 +3,7 @@ import { aimlApiProvider } from "./aimlapi";
import { alibabaCodingPlanProvider } from "./alibaba-coding-plan";
import { amazonBedrockProvider } from "./amazon-bedrock";
import { anthropicProvider } from "./anthropic";
import { azureProvider } from "./azure";
import { cerebrasProvider } from "./cerebras";
import { cloudflareAiGatewayProvider } from "./cloudflare-ai-gateway";
import { cursorProvider } from "./cursor";
@@ -68,6 +69,7 @@ import { zhipuCodingPlanProvider } from "./zhipu-coding-plan";
* list for the loginable providers; non-login model providers are appended.
*/
const ALL = [
azureProvider,
openaiCodexProvider,
anthropicProvider,
zaiProvider,
+32
View File
@@ -426,6 +426,11 @@ export interface ToolCall {
arguments: Record<string, any>;
thoughtSignature?: string; // Google-specific: opaque signature for reusing thought context
intent?: string; // Harness-level intent metadata extracted from traced tool arguments
/**
* Verbatim in-band syntax block that produced this synthetic `ptc_*` call.
* Present only for owned prompt/tool-call formats; provider-native calls omit it.
*/
rawBlock?: string;
/**
* Original wire-level name when the tool was invoked via OpenAI's custom-tool
* mechanism (e.g., `apply_patch`). Set by `openai-responses` on receive so
@@ -582,6 +587,24 @@ export type TSchema = ZodType | TJsonSchema;
/** Resolve parameter types for tool execution / handlers. */
export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: infer T } ? T : unknown;
export interface ToolCallExample<TArgs = Record<string, unknown>> {
caption?: string;
call: TArgs;
}
export interface ToolCompareExample<TArgs = Record<string, unknown>> {
caption?: string;
bad: TArgs;
good: TArgs;
}
export interface ToolNoteExample {
caption: string;
note?: string;
}
export type ToolExample<TArgs = Record<string, unknown>> =
| ToolCallExample<TArgs>
| ToolCompareExample<TArgs>
| ToolNoteExample;
export interface Tool<TParameters extends TSchema = TSchema> {
name: string;
description: string;
@@ -605,6 +628,15 @@ export interface Tool<TParameters extends TSchema = TSchema> {
* calls route correctly. Absent for regular JSON function tools.
*/
customWireName?: string;
/**
* Illustrative calls/notes; the AI layer renders them into an `<examples>`
* block in the model's native tool-call syntax and appends to the wire
* description. Author `call`/`bad`/`good` as plain argument objects WITHOUT
* `_i` — when intent tracing injects `_i` into the schema, the renderer adds
* a placeholder `_i` automatically. Type each tool's `examples` against its
* own schema (e.g. `readonly ToolExample<z.input<typeof schema>>[]`).
*/
examples?: readonly ToolExample[];
}
export interface Context {
@@ -7,7 +7,7 @@
* hashline DSL form. Other tools and surfaces fall through to
* abort-and-retry handled by the agent loop.
*/
import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai";
import type { AssistantMessage, Model, ToolCall } from "../types";
// Single source of truth for the marker pattern. `M` in the errata.
// Use a fresh non-global instance for `.test()` to avoid lastIndex pitfalls.
+1
View File
@@ -9,5 +9,6 @@ export * from "./meta-validator";
export * from "./normalize";
export * from "./spill";
export * from "./types";
export * from "./typescript";
export * from "./wire";
export * from "./zod-decontaminate";
+198
View File
@@ -0,0 +1,198 @@
/**
* Render a JSON Schema as a simplified, human-readable TypeScript type.
*
* This is a *display* conversion, not a faithful TS codegen: it surfaces the
* shape (objects, arrays, unions, enums, records) and property descriptions so
* a model — or a human reading `/dump` — can grasp a tool's parameters at a
* glance, far more legibly than raw JSON Schema. Refinement keywords
* (min/max/pattern/format) are intentionally dropped; only type structure,
* literal enums/consts, and descriptions survive.
*/
import { isJsonObject } from "./types";
export interface JsonSchemaToTsOptions {
/** Indentation unit for nested object bodies. Default two spaces. */
readonly indent?: string;
/** Emit `description` keywords as JSDoc comments on object properties. Default true. */
readonly comments?: boolean;
}
interface Ctx {
readonly indent: string;
readonly comments: boolean;
readonly defs: Record<string, unknown> | undefined;
readonly seen: Set<unknown>;
}
const SAFE_KEY = /^[A-Za-z_$][A-Za-z0-9_$]*$/;
const LOCAL_REF = /^#\/(?:\$defs|definitions)\/(.+)$/;
/** Inline an array item as `T[]` only while it stays a short single token. */
const INLINE_ARRAY_LIMIT = 40;
function literal(value: unknown): string {
if (typeof value === "string") return JSON.stringify(value);
if (typeof value === "number" || typeof value === "boolean" || value === null) return String(value);
return JSON.stringify(value) ?? "unknown";
}
/** Join member types into a TS union, deduping structurally identical renders. */
function joinUnion(parts: readonly string[]): string {
const seen = new Set<string>();
const unique: string[] = [];
for (const part of parts) {
if (seen.has(part)) continue;
seen.add(part);
unique.push(part);
}
return unique.length > 0 ? unique.join(" | ") : "never";
}
function emitJsDoc(lines: string[], description: string, pad: string): void {
// `* /` keeps a stray closing token inside the description from ending the comment.
const safe = description.replace(/\*\//g, "* /");
if (!safe.includes("\n")) {
lines.push(`${pad}/** ${safe} */`);
return;
}
lines.push(`${pad}/**`);
for (const line of safe.split("\n")) lines.push(`${pad} * ${line}`);
lines.push(`${pad} */`);
}
function convertArray(node: Record<string, unknown>, ctx: Ctx, pad: string): string {
const prefixItems = node.prefixItems;
if (Array.isArray(prefixItems)) {
return `[${prefixItems.map(item => convert(item, ctx, pad)).join(", ")}]`;
}
const items = node.items;
if (items === undefined || items === true) return "unknown[]";
if (items === false) return "never[]";
const inner = convert(items, ctx, pad);
if (inner.includes("\n") || inner.includes(" | ") || inner.length > INLINE_ARRAY_LIMIT) {
return `Array<${inner}>`;
}
return `${inner}[]`;
}
function convertObject(node: Record<string, unknown>, ctx: Ctx, pad: string): string {
const properties = isJsonObject(node.properties) ? node.properties : undefined;
const additional = node.additionalProperties;
const childPad = pad + ctx.indent;
const body: string[] = [];
if (properties) {
const required = new Set(
Array.isArray(node.required) ? node.required.filter((key): key is string => typeof key === "string") : [],
);
for (const key in properties) {
const value = properties[key];
if (
ctx.comments &&
isJsonObject(value) &&
typeof value.description === "string" &&
value.description.length > 0
) {
emitJsDoc(body, value.description, childPad);
}
const optional = required.has(key) ? "" : "?";
const name = SAFE_KEY.test(key) ? key : JSON.stringify(key);
body.push(`${childPad}${name}${optional}: ${convert(value, ctx, childPad)};`);
}
}
// No named properties: pure record / open / empty object.
if (body.length === 0) {
if (isJsonObject(additional)) return `Record<string, ${convert(additional, ctx, pad)}>`;
if (additional === true) return "Record<string, unknown>";
return "{}";
}
// Named properties alongside a free-form value schema → index signature.
if (isJsonObject(additional)) {
body.push(`${childPad}[key: string]: ${convert(additional, ctx, childPad)};`);
}
return `{\n${body.join("\n")}\n${pad}}`;
}
function convertType(type: string, node: Record<string, unknown>, ctx: Ctx, pad: string): string {
switch (type) {
case "string":
return "string";
case "integer":
case "number":
return "number";
case "boolean":
return "boolean";
case "null":
return "null";
case "array":
return convertArray(node, ctx, pad);
case "object":
return convertObject(node, ctx, pad);
default:
return "unknown";
}
}
function convert(node: unknown, ctx: Ctx, pad: string): string {
if (node === true) return "unknown";
if (node === false) return "never";
if (!isJsonObject(node)) return "unknown";
const ref = node.$ref;
if (typeof ref === "string") {
const match = LOCAL_REF.exec(ref);
const resolved = match && ctx.defs ? ctx.defs[match[1]] : undefined;
if (isJsonObject(resolved) && !ctx.seen.has(resolved)) {
ctx.seen.add(resolved);
const out = convert(resolved, ctx, pad);
ctx.seen.delete(resolved);
return out;
}
return ref.slice(ref.lastIndexOf("/") + 1);
}
if ("const" in node) return literal(node.const);
if (Array.isArray(node.enum)) {
return node.enum.length > 0 ? joinUnion(node.enum.map(literal)) : "never";
}
const union = Array.isArray(node.anyOf) ? node.anyOf : Array.isArray(node.oneOf) ? node.oneOf : undefined;
if (union) return joinUnion(union.map(variant => convert(variant, ctx, pad)));
if (Array.isArray(node.allOf)) {
return node.allOf.map(variant => convert(variant, ctx, pad)).join(" & ");
}
const type = node.type;
if (Array.isArray(type)) {
return joinUnion(type.map(entry => convertType(String(entry), node, ctx, pad)));
}
if (typeof type === "string") return convertType(type, node, ctx, pad);
return "unknown";
}
/** Convert a JSON Schema object into a simplified TypeScript type string. */
export function jsonSchemaToTypeScript(schema: unknown, options?: JsonSchemaToTsOptions): string {
const root = isJsonObject(schema) ? schema : undefined;
let defs: Record<string, unknown> | undefined;
if (root) {
for (const key of ["definitions", "$defs"] as const) {
const value = root[key];
if (isJsonObject(value)) {
defs ??= {};
Object.assign(defs, value);
}
}
}
const ctx: Ctx = {
indent: options?.indent ?? " ",
comments: options?.comments ?? true,
defs,
seen: new Set(),
};
return convert(schema, ctx, "");
}
+57 -494
View File
@@ -2,60 +2,20 @@
* Streaming-safe filters for leaked chat-template tool-call and thinking markup.
*
* Hosted models sometimes leak raw template markup into visible `content` instead
* of returning structured events. One `StreamMarkupHealing` instance owns one stream
* and one grammar selected by options:
*
* - `kimi`: Kimi K2 `<|tool_calls_section_begin|>` sections.
* - `dsml`: DeepSeek `<|DSML|tool_calls>` envelopes.
* - `thinking`: plain `<think>` / `<thinking>` blocks used by MiniMax-style streams.
*
* The parser strips marker bytes, reconstructs embedded calls, emits thinking
* deltas for thinking blocks, and holds partial tags across chunk boundaries.
* of returning structured events. Tool-call healing delegates to the same
* grammar scanners used by owned in-band tool calling; this file keeps the
* provider-facing compatibility wrapper and model/provider gating.
*/
import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity";
import { parseJsonWithRepair } from "./json-parse";
import { createInbandScanner } from "../grammar/factory";
import { ThinkingInbandScanner } from "../grammar/thinking";
import type { InbandScanEvent, InbandScanner } from "../grammar/types";
const KIMI_SECTION_BEGIN = "<|tool_calls_section_begin|>";
const KIMI_SECTION_END = "<|tool_calls_section_end|>";
const KIMI_CALL_BEGIN = "<|tool_call_begin|>";
const KIMI_CALL_END = "<|tool_call_end|>";
const KIMI_ARG_BEGIN = "<|tool_call_argument_begin|>";
const KIMI_TOKENS = [KIMI_SECTION_BEGIN, KIMI_SECTION_END, KIMI_CALL_BEGIN, KIMI_CALL_END, KIMI_ARG_BEGIN] as const;
/** Maximum buffered Kimi partial-token length before giving up holdback. */
const MAX_KIMI_PARTIAL_HOLD = 64;
/** Both fullwidth (U+FF5C) and ASCII pipes are observed in DeepSeek DSML leaks. */
const DSML_PIPE = "[||]";
const DSML_TOOL_CALLS_OPEN_RE = new RegExp(`<${DSML_PIPE}DSML${DSML_PIPE}tool_calls>`, "y");
const DSML_TOOL_CALLS_CLOSE_RE = new RegExp(`</${DSML_PIPE}DSML${DSML_PIPE}tool_calls>`, "y");
const DSML_INVOKE_OPEN_RE = new RegExp(`<${DSML_PIPE}DSML${DSML_PIPE}invoke\\s+name="([^"]*)"\\s*>`, "y");
const DSML_INVOKE_CLOSE_RE = new RegExp(`</${DSML_PIPE}DSML${DSML_PIPE}invoke>`, "y");
const DSML_PARAMETER_OPEN_RE = new RegExp(
`<${DSML_PIPE}DSML${DSML_PIPE}parameter\\s+name="([^"]*)"(?:\\s+string="(true|false)")?\\s*>`,
"y",
);
const DSML_PARAMETER_CLOSE_RE = new RegExp(`</${DSML_PIPE}DSML${DSML_PIPE}parameter>`, "y");
/** Canonical DSML section-open shape; `|` positions accept either pipe variant. */
const DSML_SECTION_OPEN_TEMPLATE = "<|DSML|tool_calls>";
const THINK_OPEN = "<think>";
const THINK_CLOSE = "</think>";
const THINKING_OPEN = "<thinking>";
const THINKING_CLOSE = "</thinking>";
const PLAIN_THINKING_TAGS = [
{ open: THINK_OPEN, close: THINK_CLOSE },
{ open: THINKING_OPEN, close: THINKING_CLOSE },
] as const;
/** Cap held-back XML tag bytes so a stray `<` in prose cannot grow unboundedly. */
const MAX_XML_PARTIAL_HOLD = 256;
/** Maximum parameter bytes to accumulate before abandoning a pathological XML call. */
const MAX_XML_PARAM_VALUE_LENGTH = 1_000_000;
const DSML_TOOL_CALLS_CLOSE_FULLWIDTH = "</|DSML|tool_calls>";
const DSML_TOOL_CALLS_CLOSE_ASCII = "</|DSML|tool_calls>";
export interface HealedToolCall {
readonly id: string;
@@ -74,48 +34,28 @@ export type StreamMarkupHealingEvent =
| { readonly type: "thinking"; readonly thinking: string }
| { readonly type: "toolCall"; readonly call: HealedToolCall };
type XmlToolState =
| { readonly kind: "idle" }
| { readonly kind: "section" }
| { readonly kind: "invoke"; readonly name: string; readonly args: Record<string, unknown> }
| {
readonly kind: "parameter";
readonly invokeName: string;
readonly args: Record<string, unknown>;
readonly paramName: string;
readonly isString: boolean;
value: string;
truncated?: boolean;
};
type ThinkingTag = { readonly open: string; readonly close: string };
/**
* State machine that consumes streamed visible text and emits cleaned text,
* thinking deltas, and reconstructed tool calls.
*
* Feed only one stream channel (usually `delta.content` / `message.content`).
* Mixing reasoning and visible text into the same instance can corrupt the
* held-back partial tag buffer.
* Mixing reasoning and visible text into the same instance can corrupt held-back
* partial tag buffers.
*/
export class StreamMarkupHealing {
readonly #pattern: StreamMarkupHealingPattern;
#buffer = "";
#offset = 0;
#kimiInSection = false;
#kimiInCall = false;
#kimiInArgs = false;
#kimiPendingId = "";
#kimiPendingArgs = "";
#xmlState: XmlToolState = { kind: "idle" };
#thinkingCloseTag = "";
readonly #scanner: InbandScanner;
#sectionTerminated = false;
readonly #completed: HealedToolCall[] = [];
constructor(options: StreamMarkupHealingOptions) {
this.#pattern = options.pattern;
this.#scanner =
options.pattern === "kimi"
? createInbandScanner("kimi")
: options.pattern === "dsml"
? createInbandScanner("xml", { xmlTagset: "dsml" })
: new ThinkingInbandScanner();
}
get pattern(): StreamMarkupHealingPattern {
@@ -125,8 +65,8 @@ export class StreamMarkupHealing {
/**
* Feed a chunk and return visible text only. Reconstructed tool calls are
* stored for {@link drainCompleted}; thinking blocks are intentionally not
* returned by this compatibility helper. Use {@link feedEvents} when the
* caller needs ordered text/thinking/tool-call events.
* returned by this compatibility helper. Use {@link feedEvents} when the caller
* needs ordered text/thinking/tool-call events.
*/
feed(text: string): string {
let clean = "";
@@ -143,16 +83,8 @@ export class StreamMarkupHealing {
/** Feed a chunk and return cleaned text/thinking/tool-call events in stream order. */
feedEvents(text: string): StreamMarkupHealingEvent[] {
if (text.length === 0) return [];
this.#compact();
this.#buffer += text;
switch (this.#pattern) {
case "kimi":
return this.#consumeKimiEvents();
case "dsml":
return this.#consumeDsmlEvents();
case "thinking":
return this.#consumePlainThinkingEvents();
}
this.#markSectionClosed(text);
return this.#convertScannerEvents(this.#scanner.feed(text));
}
/**
@@ -176,32 +108,12 @@ export class StreamMarkupHealing {
/**
* Flush held-back stream-end fragments as ordered events. Partial tool-call
* sections/envelopes are dropped; unterminated thinking blocks are emitted as
* thinking, matching the previous MiniMax parser behavior.
* sections/envelopes are dropped by the delegated scanners; unterminated
* thinking blocks are emitted as thinking, matching the previous MiniMax parser
* behavior.
*/
flushEvents(): StreamMarkupHealingEvent[] {
const tail = this.#remaining();
this.#buffer = "";
this.#offset = 0;
switch (this.#pattern) {
case "kimi": {
const inTemplate = this.#kimiInCall || this.#kimiInSection;
this.#resetKimi();
return inTemplate || tail.length === 0 ? [] : [{ type: "text", text: tail }];
}
case "dsml": {
const state = this.#xmlState;
this.#xmlState = { kind: "idle" };
return state.kind !== "idle" || tail.length === 0 ? [] : [{ type: "text", text: tail }];
}
case "thinking": {
const closeTag = this.#thinkingCloseTag;
this.#thinkingCloseTag = "";
if (tail.length === 0) return [];
return closeTag ? [{ type: "thinking", thinking: tail }] : [{ type: "text", text: tail }];
}
}
return this.#convertScannerEvents(this.#scanner.flush());
}
/** Flush held-back text only. Reconstructed calls are retained for {@link drainCompleted}. */
@@ -222,393 +134,44 @@ export class StreamMarkupHealing {
return this.#sectionTerminated;
}
#remaining(): string {
return this.#offset === 0 ? this.#buffer : this.#buffer.slice(this.#offset);
}
#compact(): void {
if (this.#offset === 0) return;
this.#buffer = this.#buffer.slice(this.#offset);
this.#offset = 0;
}
#consumeKimiEvents(): StreamMarkupHealingEvent[] {
const events: StreamMarkupHealingEvent[] = [];
let clean = "";
const flushClean = (): void => {
if (clean.length === 0) return;
events.push({ type: "text", text: clean });
clean = "";
};
while (this.#offset < this.#buffer.length) {
if (this.#startsWithPartialToken(KIMI_TOKENS, MAX_KIMI_PARTIAL_HOLD)) break;
if (this.#matchesToken(KIMI_SECTION_BEGIN)) {
this.#kimiInSection = true;
this.#offset += KIMI_SECTION_BEGIN.length;
continue;
}
if (this.#matchesToken(KIMI_SECTION_END)) {
this.#kimiInSection = false;
this.#sectionTerminated = true;
this.#offset += KIMI_SECTION_END.length;
continue;
}
if (this.#matchesToken(KIMI_CALL_BEGIN)) {
if (!this.#kimiInSection) {
clean += KIMI_CALL_BEGIN;
this.#offset += KIMI_CALL_BEGIN.length;
continue;
}
this.#kimiInCall = true;
this.#kimiInArgs = false;
this.#kimiPendingId = "";
this.#kimiPendingArgs = "";
this.#offset += KIMI_CALL_BEGIN.length;
continue;
}
if (this.#matchesToken(KIMI_ARG_BEGIN)) {
if (!this.#kimiInSection) {
clean += KIMI_ARG_BEGIN;
this.#offset += KIMI_ARG_BEGIN.length;
continue;
}
this.#kimiInArgs = true;
this.#offset += KIMI_ARG_BEGIN.length;
continue;
}
if (this.#matchesToken(KIMI_CALL_END)) {
if (!this.#kimiInSection || !this.#kimiInCall) {
clean += KIMI_CALL_END;
this.#offset += KIMI_CALL_END.length;
continue;
}
const call = this.#finalizeKimiCall();
flushClean();
events.push({ type: "toolCall", call });
this.#offset += KIMI_CALL_END.length;
continue;
}
const ch = this.#buffer[this.#offset]!;
this.#offset += 1;
if (this.#kimiInCall) {
if (this.#kimiInArgs) {
this.#kimiPendingArgs += ch;
} else {
this.#kimiPendingId += ch;
}
continue;
}
if (!this.#kimiInSection) clean += ch;
#markSectionClosed(text: string): void {
if (this.#sectionTerminated) return;
if (this.#pattern === "kimi") {
this.#sectionTerminated = text.includes(KIMI_SECTION_END);
return;
}
flushClean();
return events;
this.#sectionTerminated =
text.includes(DSML_TOOL_CALLS_CLOSE_FULLWIDTH) || text.includes(DSML_TOOL_CALLS_CLOSE_ASCII);
}
#consumeDsmlEvents(): StreamMarkupHealingEvent[] {
return this.#consumeXmlToolEvents({
getState: () => this.#xmlState,
setState: state => {
this.#xmlState = state;
},
sectionOpen: DSML_TOOL_CALLS_OPEN_RE,
sectionClose: DSML_TOOL_CALLS_CLOSE_RE,
invokeOpen: DSML_INVOKE_OPEN_RE,
invokeClose: DSML_INVOKE_CLOSE_RE,
parameterOpen: DSML_PARAMETER_OPEN_RE,
parameterClose: DSML_PARAMETER_CLOSE_RE,
coerceStringByDefault: true,
});
}
#consumePlainThinkingEvents(): StreamMarkupHealingEvent[] {
const events: StreamMarkupHealingEvent[] = [];
let clean = "";
let thinking = "";
const flushClean = (): void => {
if (clean.length === 0) return;
events.push({ type: "text", text: clean });
clean = "";
};
const flushThinking = (): void => {
if (thinking.length === 0) return;
events.push({ type: "thinking", thinking });
thinking = "";
};
while (this.#offset < this.#buffer.length) {
if (this.#thinkingCloseTag) {
if (this.#matchesToken(this.#thinkingCloseTag)) {
flushThinking();
this.#offset += this.#thinkingCloseTag.length;
this.#thinkingCloseTag = "";
continue;
}
if (this.#startsWithPartialToken([this.#thinkingCloseTag], MAX_XML_PARTIAL_HOLD)) break;
const ch = this.#buffer[this.#offset]!;
this.#offset += 1;
thinking += ch;
continue;
}
const thinkingTag = this.#tryMatchThinkingOpen(PLAIN_THINKING_TAGS);
if (thinkingTag) {
flushClean();
this.#thinkingCloseTag = thinkingTag.close;
continue;
}
if (this.#startsWithPartialThinkingOpen(PLAIN_THINKING_TAGS)) break;
const ch = this.#buffer[this.#offset]!;
this.#offset += 1;
clean += ch;
}
flushClean();
flushThinking();
return events;
}
#consumeXmlToolEvents(config: {
readonly getState: () => XmlToolState;
readonly setState: (state: XmlToolState) => void;
readonly sectionOpen: RegExp;
readonly sectionClose: RegExp;
readonly invokeOpen: RegExp;
readonly invokeClose: RegExp;
readonly parameterOpen: RegExp;
readonly parameterClose: RegExp;
readonly coerceStringByDefault: boolean;
}): StreamMarkupHealingEvent[] {
const events: StreamMarkupHealingEvent[] = [];
let clean = "";
const flushClean = (): void => {
if (clean.length === 0) return;
events.push({ type: "text", text: clean });
clean = "";
};
while (this.#offset < this.#buffer.length) {
const state = config.getState();
if (state.kind === "idle") {
if (this.#tryMatch(config.sectionOpen)) {
config.setState({ kind: "section" });
continue;
}
} else if (state.kind === "section") {
if (this.#tryMatch(config.sectionClose)) {
config.setState({ kind: "idle" });
this.#sectionTerminated = true;
continue;
}
const invokeMatch = this.#tryMatchCapture(config.invokeOpen);
if (invokeMatch) {
config.setState({ kind: "invoke", name: invokeMatch[1] ?? "", args: {} });
continue;
}
} else if (state.kind === "invoke") {
if (this.#tryMatch(config.invokeClose)) {
const call = finalizeXmlToolCall(state.name, state.args);
flushClean();
events.push({ type: "toolCall", call });
config.setState({ kind: "section" });
continue;
}
const paramMatch = this.#tryMatchCapture(config.parameterOpen);
if (paramMatch) {
const stringAttr = paramMatch[2];
config.setState({
kind: "parameter",
invokeName: state.name,
args: state.args,
paramName: paramMatch[1] ?? "",
isString: config.coerceStringByDefault ? stringAttr !== "false" : false,
value: "",
#convertScannerEvents(events: readonly InbandScanEvent[]): StreamMarkupHealingEvent[] {
const out: StreamMarkupHealingEvent[] = [];
for (const event of events) {
switch (event.type) {
case "text":
out.push({ type: "text", text: event.text });
break;
case "thinkingDelta":
if (event.delta.length > 0) out.push({ type: "thinking", thinking: event.delta });
break;
case "toolEnd":
out.push({
type: "toolCall",
call: {
id: generateHealedToolCallId(),
name: event.name,
arguments: JSON.stringify(event.arguments),
},
});
continue;
}
} else if (this.#tryMatch(config.parameterClose)) {
// A capped value executes with silently corrupted input unless the
// truncation is made explicit — the marker fails JSON params loudly
// and tells the model/tool what happened to string params.
const paramValue = state.truncated
? `${state.value}\n…[parameter truncated: exceeded ${MAX_XML_PARAM_VALUE_LENGTH} bytes]`
: state.value;
state.args[state.paramName] = coerceXmlParamValue(paramValue, state.isString);
config.setState({ kind: "invoke", name: state.invokeName, args: state.args });
continue;
}
if (state.kind === "idle") {
// In idle, a bare `<` is legitimate output (`a < b`, generics, JSX).
// Only hold back tails that could still grow into the DSML
// section-open tag; everything else flows through immediately.
if (this.#startsWithPartialDsmlSectionOpen()) break;
} else if (this.#startsWithPartialXmlTag()) {
break;
}
const ch = this.#buffer[this.#offset]!;
this.#offset += 1;
if (state.kind === "idle") {
clean += ch;
continue;
}
if (state.kind === "parameter") {
if (state.value.length < MAX_XML_PARAM_VALUE_LENGTH) {
state.value += ch;
} else {
// Beyond the cap the value stops growing, but we stay in
// `parameter` state so the rest of the envelope — including its
// close tags — is still swallowed instead of leaking into
// visible text. The close handler appends an explicit marker.
state.truncated = true;
}
break;
case "thinkingStart":
case "thinkingEnd":
case "toolStart":
case "toolArgDelta":
break;
}
}
flushClean();
return events;
}
#tryMatch(pattern: RegExp): boolean {
pattern.lastIndex = this.#offset;
const match = pattern.exec(this.#buffer);
if (!match) return false;
this.#offset += match[0].length;
return true;
}
#tryMatchCapture(pattern: RegExp): RegExpExecArray | undefined {
pattern.lastIndex = this.#offset;
const match = pattern.exec(this.#buffer);
if (!match) return undefined;
this.#offset += match[0].length;
return match;
}
#tryMatchThinkingOpen(tags: readonly ThinkingTag[]): ThinkingTag | undefined {
for (const tag of tags) {
if (!this.#matchesToken(tag.open)) continue;
this.#offset += tag.open.length;
return tag;
}
return undefined;
}
#matchesToken(token: string): boolean {
return this.#buffer.startsWith(token, this.#offset);
}
#startsWithPartialThinkingOpen(tags: readonly ThinkingTag[]): boolean {
for (const tag of tags) {
if (this.#startsWithPartialToken([tag.open], MAX_XML_PARTIAL_HOLD)) return true;
}
return false;
}
#startsWithPartialToken(tokens: readonly string[], maxHold: number): boolean {
const remainingLength = this.#buffer.length - this.#offset;
if (remainingLength === 0 || remainingLength > maxHold) return false;
for (const token of tokens) {
if (token.length <= remainingLength) continue;
if (this.#bufferIsPrefixOf(token, remainingLength)) return true;
}
return false;
}
#startsWithPartialXmlTag(): boolean {
if (this.#buffer[this.#offset] !== "<") return false;
const tailLength = this.#buffer.length - this.#offset;
if (tailLength > MAX_XML_PARTIAL_HOLD) return false;
for (let i = this.#offset + 1; i < this.#buffer.length; i++) {
if (this.#buffer[i] === ">") return false;
}
return true;
}
#startsWithPartialDsmlSectionOpen(): boolean {
const tailLength = this.#buffer.length - this.#offset;
if (tailLength === 0 || tailLength >= DSML_SECTION_OPEN_TEMPLATE.length) return false;
for (let i = 0; i < tailLength; i++) {
const ch = this.#buffer[this.#offset + i]!;
const expected = DSML_SECTION_OPEN_TEMPLATE[i]!;
if (expected === "|") {
if (ch !== "|" && ch !== "|") return false;
} else if (ch !== expected) {
return false;
}
}
return true;
}
#bufferIsPrefixOf(token: string, remainingLength: number): boolean {
for (let i = 0; i < remainingLength; i++) {
if (this.#buffer[this.#offset + i] !== token[i]) return false;
}
return true;
}
#finalizeKimiCall(): HealedToolCall {
const rawId = this.#kimiPendingId.trim();
const rawArgs = this.#kimiPendingArgs.trim();
const name = normalizeKimiFunctionName(rawId);
let argsJson = rawArgs;
if (rawArgs.length > 0) {
try {
argsJson = JSON.stringify(parseJsonWithRepair<unknown>(rawArgs));
} catch {
// Leave raw; downstream parseStreamingJson absorbs the failure.
}
} else {
argsJson = "{}";
}
this.#kimiInCall = false;
this.#kimiInArgs = false;
this.#kimiPendingId = "";
this.#kimiPendingArgs = "";
return { id: generateHealedToolCallId(), name, arguments: argsJson };
}
#resetKimi(): void {
this.#kimiInSection = false;
this.#kimiInCall = false;
this.#kimiInArgs = false;
this.#kimiPendingId = "";
this.#kimiPendingArgs = "";
}
}
function normalizeKimiFunctionName(rawId: string): string {
const stripped = rawId.startsWith("functions.") ? rawId.slice("functions.".length) : rawId;
const colon = stripped.indexOf(":");
return colon >= 0 ? stripped.slice(0, colon) : stripped;
}
function finalizeXmlToolCall(name: string, args: Record<string, unknown>): HealedToolCall {
return {
id: generateHealedToolCallId(),
name: name.trim(),
arguments: JSON.stringify(args),
};
}
function coerceXmlParamValue(raw: string, isString: boolean): unknown {
if (isString) return raw;
const trimmed = raw.trim();
if (trimmed.length === 0) return raw;
try {
return parseJsonWithRepair<unknown>(trimmed);
} catch {
return raw;
return out;
}
}
+98 -22
View File
@@ -11,9 +11,9 @@
* 2. Normalizes LLM quirks (null / "null" → omit-or-default substitution)
* against the JSON Schema before validation.
* 3. Validates with the Zod or JSON-Schema validator.
* 4. On failure, walks the resulting issues and coerces JSON-stringified
* values (`"[1,2]"` → `[1,2]`), drops unrecognized keys, and retries up
* to `MAX_COERCION_PASSES` times.
* 4. On failure, walks the resulting issues and coerces common LLM type
* drift (JSON-stringified values, boolean/number/string scalar drift),
* drops unrecognized keys, and retries up to `MAX_COERCION_PASSES` times.
* 5. Throws a formatted error if reconciliation fails; otherwise returns
* the parsed arguments with original unknown root fields preserved (so
* hallucinated top-level keys still surface to the caller).
@@ -38,20 +38,20 @@ import { isZodSchema, zodToWireSchema } from "./schema/wire";
// Type Coercion Utilities
// ============================================================================
//
// LLMs sometimes produce tool arguments where a value that should be a number,
// boolean, array, or object is instead passed as a JSON-encoded string. For
// example, an array parameter might arrive as `"[1, 2, 3]"` instead of `[1, 2, 3]`.
// LLMs sometimes produce tool arguments where a value has the right meaning but
// the wrong JSON type. For example, an array parameter might arrive as
// `"[1, 2, 3]"`, a boolean as `"yes"` or `1`, or a string field as a structured
// object that should be embedded verbatim.
//
// Rather than rejecting these outright, we attempt automatic coercion:
// 1. Validate against the tool's schema (Zod, derived from TypeBox when the
// tool was authored with TypeBox).
// 2. For each type error where the actual value is a string, we check if
// parsing it as JSON yields a value matching the expected type.
// 3. If so, we replace the string with the parsed value and re-validate.
// 2. For each type error, perform only the schema-directed rewrite that
// matches the expected type.
// 3. Re-validate the full argument object after each coercion pass.
//
// This is intentionally conservative: we only parse strings that look like
// valid JSON literals (objects, arrays, booleans, null, numbers) and only
// accept the result if it matches the schema's expected type.
// This is intentionally conservative: each rewrite is small and validation
// remains the source of truth for whether the result is accepted.
// ============================================================================
/** Regex matching valid JSON number literals (integers, decimals, scientific notation) */
@@ -109,6 +109,85 @@ function tryParseNumberString(value: string, expectedTypes: string[]): { value:
return { value: parsed, changed: true };
}
function tryCoerceBoolean(value: unknown, expectedTypes: string[]): { value: unknown; changed: boolean } {
if (!expectedTypes.includes("boolean")) {
return { value, changed: false };
}
if (typeof value === "number") {
if (value === 0) return { value: false, changed: true };
if (value === 1) return { value: true, changed: true };
return { value, changed: false };
}
if (typeof value !== "string") {
return { value, changed: false };
}
switch (value.trim().toLowerCase()) {
case "true":
case "1":
case "yes":
case "on":
return { value: true, changed: true };
case "false":
case "0":
case "no":
case "off":
return { value: false, changed: true };
default:
return { value, changed: false };
}
}
function tryCoerceBooleanToNumber(value: unknown, expectedTypes: string[]): { value: unknown; changed: boolean } {
if (!expectedTypes.includes("number") && !expectedTypes.includes("integer")) {
return { value, changed: false };
}
if (typeof value !== "boolean") {
return { value, changed: false };
}
return { value: value ? 1 : 0, changed: true };
}
function tryCoerceString(value: unknown, expectedTypes: string[]): { value: unknown; changed: boolean } {
if (!expectedTypes.includes("string") || typeof value === "string" || value === null || value === undefined) {
return { value, changed: false };
}
if (Array.isArray(value) || typeof value === "object") {
try {
const stringified = JSON.stringify(value);
if (stringified === undefined) return { value, changed: false };
return { value: stringified, changed: true };
} catch {
return { value, changed: false };
}
}
if (typeof value === "function") {
return { value, changed: false };
}
return { value: String(value), changed: true };
}
function tryCoerceForExpectedTypes(value: unknown, expectedTypes: string[]): { value: unknown; changed: boolean } {
if (typeof value === "string") {
const parsed = tryParseJsonForTypes(value, expectedTypes);
if (parsed.changed) return parsed;
return tryCoerceBoolean(value, expectedTypes);
}
const booleanCoercion = tryCoerceBoolean(value, expectedTypes);
if (booleanCoercion.changed) return booleanCoercion;
const numericCoercion = tryCoerceBooleanToNumber(value, expectedTypes);
if (numericCoercion.changed) return numericCoercion;
return tryCoerceString(value, expectedTypes);
}
function tryParseLeadingJsonContainer(value: string): unknown | undefined {
const firstChar = value[0];
const closingChar = firstChar === "{" ? "}" : firstChar === "[" ? "]" : undefined;
@@ -806,10 +885,11 @@ function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
* Repair issues raised by the validator before we surface them to the caller.
*
* Two kinds of repair are applied:
* - **type**: when a value is a JSON-encoded string and the schema wants
* something else, parse it and substitute the parsed value. When a
* non-union schema wants an array but receives a singleton value, wrap that
* value in a one-element array.
* - **type**: when a value has a common LLM-produced shape mismatch, rewrite
* it only in the direction requested by the schema: parse JSON strings,
* accept boolean spellings, stringify non-null values for string fields,
* map booleans to numeric 0/1, and wrap singleton array values for non-union
* array expectations.
* - **unrecognized**: when a strict object received an extra key (Zod's
* `unrecognized_keys` or JSON Schema's `additionalProperties: false`),
* drop that key so re-validation succeeds. This effectively coerces every
@@ -818,9 +898,8 @@ function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
*
* The function is safe and conservative:
* - Only processes "type" and "unrecognized" issues
* - Only attempts JSON coercion on string values
* - Only attempts schema-directed coercions for the expected type
* - Only wraps singleton array values for non-union type expectations
* - Only accepts parsed results that match the expected type
* - Clones the args object before mutation (copy-on-write)
*/
function coerceArgsFromIssues(args: unknown, issues: FlatIssue[]): { value: unknown; changed: boolean } {
@@ -845,10 +924,7 @@ function coerceArgsFromIssues(args: unknown, issues: FlatIssue[]): { value: unkn
if (issue.expectedTypes.length === 0) continue;
const currentValue = getValueAtPointer(nextArgs, issue.instancePath);
const result =
typeof currentValue === "string"
? tryParseJsonForTypes(currentValue, issue.expectedTypes)
: { value: currentValue, changed: false };
const result = tryCoerceForExpectedTypes(currentValue, issue.expectedTypes);
const coercedValue = result.changed
? result.value
: issue.expectedTypes.includes("array") &&
@@ -0,0 +1,244 @@
import { describe, expect, it } from "bun:test";
import {
type BlockState,
mergeCursorMcpToolCallArgs,
processInteractionUpdate,
type ToolCallState,
type UsageState,
} from "@oh-my-pi/pi-ai/providers/cursor";
import type { AssistantMessage, AssistantMessageEvent, TextContent, ThinkingContent } from "@oh-my-pi/pi-ai/types";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
interface Harness {
output: AssistantMessage;
stream: AssistantMessageEventStream;
captured: AssistantMessageEvent[];
state: BlockState;
usageState: UsageState;
}
function newHarness(): Harness {
const output: AssistantMessage = {
role: "assistant",
content: [],
api: "cursor-agent",
provider: "cursor",
model: "cursor-composer-2.5",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 0,
};
const stream = new AssistantMessageEventStream();
const captured: AssistantMessageEvent[] = [];
const origPush = stream.push.bind(stream);
stream.push = (event: AssistantMessageEvent) => {
captured.push(event);
origPush(event);
};
let textBlock: (TextContent & { index: number }) | null = null;
let thinkingBlock: (ThinkingContent & { index: number }) | null = null;
let toolCall: ToolCallState | null = null;
const state: BlockState = {
get currentTextBlock() {
return textBlock;
},
get currentThinkingBlock() {
return thinkingBlock;
},
get currentToolCall() {
return toolCall;
},
firstTokenTime: undefined,
setTextBlock: b => {
textBlock = b;
},
setThinkingBlock: b => {
thinkingBlock = b;
},
setToolCall: t => {
toolCall = t;
},
setFirstTokenTime: () => {},
};
return { output, stream, captured, state, usageState: { sawTokenDelta: false } };
}
function startMcpToolCall(h: Harness, name: string, id = "call-1"): void {
processInteractionUpdate(
{
message: {
case: "toolCallStarted",
value: {
callId: id,
toolCall: {
mcpToolCall: { args: { name, toolName: name, toolCallId: id } },
},
},
},
},
h.output,
h.stream,
h.state,
h.usageState,
);
}
function pushArgsTextDelta(h: Harness, argsTextDelta: string): void {
processInteractionUpdate(
{ message: { case: "partialToolCall", value: { argsTextDelta } } },
h.output,
h.stream,
h.state,
h.usageState,
);
}
function completeMcpToolCall(h: Harness, args: Record<string, Uint8Array> | undefined): void {
processInteractionUpdate(
{
message: {
case: "toolCallCompleted",
value: { toolCall: { mcpToolCall: { args: { args } } } },
},
},
h.output,
h.stream,
h.state,
h.usageState,
);
}
describe("mergeCursorMcpToolCallArgs", () => {
it("returns streamed args unchanged when completion is undefined", () => {
const streamed = { tasks: [{ assignment: "do" }], context: "ctx" };
expect(mergeCursorMcpToolCallArgs(streamed, undefined)).toEqual(streamed);
});
it("preserves streamed keys the completion frame omits", () => {
// Issue #2615: the completion frame's McpArgs map drops oversized
// parameters. The task tool's `tasks` array was being lost when only
// the smaller `context` key survived the completion frame.
const streamed = { tasks: [{ assignment: "do A" }, { assignment: "do B" }], context: "ctx" };
const completion = { context: "ctx" };
expect(mergeCursorMcpToolCallArgs(streamed, completion)).toEqual({
tasks: [{ assignment: "do A" }, { assignment: "do B" }],
context: "ctx",
});
});
it("adopts scalar values from the completion frame when present", () => {
const streamed = { agent: "task", context: "partial" };
const completion = { agent: "task", context: "final" };
expect(mergeCursorMcpToolCallArgs(streamed, completion)).toEqual({ agent: "task", context: "final" });
});
it("keeps the streamed structured value when completion downgrades to a raw string", () => {
// decodeMcpArgValue returns the raw decoded string when the byte payload
// cannot be parsed as JSON. The streamed JSON is structurally richer, so
// merge must prefer it over the string fallback.
const streamed = { tasks: [{ assignment: "do A" }] };
const completion = { tasks: "[{assignment: 'do A'}]" };
expect(mergeCursorMcpToolCallArgs(streamed, completion)).toEqual({ tasks: [{ assignment: "do A" }] });
});
it("accepts completion-only keys that the streamed args never carried", () => {
const streamed = { agent: "task" };
const completion = { agent: "task", model: "default" };
expect(mergeCursorMcpToolCallArgs(streamed, completion)).toEqual({ agent: "task", model: "default" });
});
it("returns an empty object when both sides are absent", () => {
expect(mergeCursorMcpToolCallArgs(undefined, undefined)).toEqual({});
});
});
describe("processInteractionUpdate args_text_delta handling", () => {
it("treats cumulative argsTextDelta snapshots as snapshots, not append-only fragments", () => {
const h = newHarness();
startMcpToolCall(h, "task");
// Cursor emits aggregated args text so far on each delta.
const cumulative = [
`{"agent":"task","tas`,
`{"agent":"task","tasks":[{"assignme`,
`{"agent":"task","tasks":[{"assignment":"do A"},{"assignment":"do B"}]}`,
];
for (const snapshot of cumulative) {
pushArgsTextDelta(h, snapshot);
}
const block = h.state.currentToolCall!;
expect(block.partialJson).toBe(cumulative[cumulative.length - 1]);
expect(block.arguments).toEqual({
agent: "task",
tasks: [{ assignment: "do A" }, { assignment: "do B" }],
});
// Each cumulative snapshot only emits the new suffix as the delta event.
const deltas = h.captured.filter(e => e.type === "toolcall_delta").map(e => (e as { delta: string }).delta);
expect(deltas.join("")).toBe(cumulative[cumulative.length - 1]);
expect(deltas).toEqual([`{"agent":"task","tas`, `ks":[{"assignme`, `nt":"do A"},{"assignment":"do B"}]}`]);
});
it("still appends genuinely incremental argsTextDelta fragments", () => {
const h = newHarness();
startMcpToolCall(h, "task");
const fragments = [`{"agent":`, `"task",`, `"items":[1,2,3]}`];
for (const fragment of fragments) {
pushArgsTextDelta(h, fragment);
}
expect(h.state.currentToolCall!.partialJson).toBe(fragments.join(""));
expect(h.state.currentToolCall!.arguments).toEqual({ agent: "task", items: [1, 2, 3] });
});
it("skips empty argsTextDelta snapshots without emitting a delta event", () => {
const h = newHarness();
startMcpToolCall(h, "task");
pushArgsTextDelta(h, `{"agent":"task"}`);
pushArgsTextDelta(h, `{"agent":"task"}`);
pushArgsTextDelta(h, "");
expect(h.state.currentToolCall!.partialJson).toBe(`{"agent":"task"}`);
const deltas = h.captured.filter(e => e.type === "toolcall_delta");
expect(deltas).toHaveLength(1);
});
it("preserves the streamed tasks array when the completion frame omits it (issue #2615)", () => {
const h = newHarness();
startMcpToolCall(h, "task");
const fullArgs = `{"agent":"task","tasks":[{"assignment":"do A"},{"assignment":"do B"}],"context":"ctx"}`;
// Multiple cumulative snapshots to ensure the delta path is exercised.
pushArgsTextDelta(h, fullArgs.slice(0, 30));
pushArgsTextDelta(h, fullArgs.slice(0, 60));
pushArgsTextDelta(h, fullArgs);
// Completion frame's McpArgs map omits the oversized `tasks` key but
// still carries the smaller scalars.
completeMcpToolCall(h, {
agent: new TextEncoder().encode(`"task"`),
context: new TextEncoder().encode(`"ctx"`),
});
expect(h.state.currentToolCall).toBeNull();
const finalBlock = h.output.content[0];
expect(finalBlock?.type).toBe("toolCall");
if (finalBlock?.type !== "toolCall") throw new Error("expected toolCall block");
expect(finalBlock.arguments).toEqual({
agent: "task",
tasks: [{ assignment: "do A" }, { assignment: "do B" }],
context: "ctx",
});
});
});
@@ -0,0 +1,203 @@
import { describe, expect, it } from "bun:test";
import type { ToolCall } from "@oh-my-pi/pi-ai";
import {
createInbandScanner,
getInbandGrammar,
type InbandScanEvent,
type ToolCallSyntax,
} from "@oh-my-pi/pi-ai/grammar";
function scan(syntax: ToolCallSyntax, text: string, charByChar = false): InbandScanEvent[] {
const scanner = createInbandScanner(syntax);
const events: InbandScanEvent[] = [];
if (charByChar) for (const ch of text) events.push(...scanner.feed(ch));
else events.push(...scanner.feed(text));
events.push(...scanner.flush());
return events;
}
function parsedCalls(
syntax: ToolCallSyntax,
text: string,
charByChar = false,
): { name: string; arguments: Record<string, unknown> }[] {
return scan(syntax, text, charByChar)
.filter((event): event is Extract<InbandScanEvent, { type: "toolEnd" }> => event.type === "toolEnd")
.map(event => ({ name: event.name, arguments: event.arguments }));
}
function visibleText(events: readonly InbandScanEvent[]): string {
return events
.filter((event): event is Extract<InbandScanEvent, { type: "text" }> => event.type === "text")
.map(event => event.text)
.join("");
}
const call = (name: string, args: Record<string, unknown>): ToolCall => ({
type: "toolCall",
id: name,
name,
arguments: args,
});
describe("gemini grammar (Pythonic tool_code)", () => {
it("parses the print(default_api...) form", () => {
const calls = parsedCalls("gemini", "```tool_code\nprint(default_api.read(path='a.ts', count=2))\n```");
expect(calls).toEqual([{ name: "read", arguments: { path: "a.ts", count: 2 } }]);
});
it("parses bare default_api calls and the assignment form", () => {
expect(parsedCalls("gemini", '```tool_code\ndefault_api.search(pattern="x")\n```')).toEqual([
{ name: "search", arguments: { pattern: "x" } },
]);
expect(parsedCalls("gemini", '```tool_code\nresult = search(pattern="x")\n```')).toEqual([
{ name: "search", arguments: { pattern: "x" } },
]);
});
it("decodes Python literals (bool/None/number/list/dict)", () => {
const calls = parsedCalls(
"gemini",
'```tool_code\ndefault_api.f(s="hi", n=3, r=1.5, b=True, z=None, arr=[1, 2], obj={"k": "v"})\n```',
);
expect(calls[0]!.arguments).toEqual({ s: "hi", n: 3, r: 1.5, b: true, z: null, arr: [1, 2], obj: { k: "v" } });
});
it("ignores parens and commas inside string arguments", () => {
const calls = parsedCalls("gemini", '```tool_code\ndefault_api.search(pattern="foo(a, b)", flag=False)\n```');
expect(calls[0]!.arguments).toEqual({ pattern: "foo(a, b)", flag: false });
});
it("ignores Python comments and decodes raw/unicode string literals", () => {
const text = [
"```tool_code",
'# default_api.write(path="ignored")',
"result = read(",
' path=r"src/(foo)\\.ts", # default_api.write(path="ignored")',
" count=2,",
' meta={"emoji": "\\U0001F600"},',
")",
'[default_api.write(path="out", content="foo(,bar")]',
"```",
].join("\n");
const calls = parsedCalls("gemini", text);
expect(calls.map(parsed => parsed.name)).toEqual(["read", "write"]);
expect(calls[0]!.arguments).toEqual({ path: "src/(foo)\\.ts", count: 2, meta: { emoji: "😀" } });
expect(calls[1]!.arguments).toEqual({ path: "out", content: "foo(,bar" });
});
it("parses parallel calls written as a [a, b] list", () => {
const calls = parsedCalls(
"gemini",
'```tool_code\n[default_api.read(path="a"), default_api.write(path="b", content="c")]\n```',
);
expect(calls).toEqual([
{ name: "read", arguments: { path: "a" } },
{ name: "write", arguments: { path: "b", content: "c" } },
]);
});
it("preserves prose outside the fence", () => {
const text = visibleText(scan("gemini", 'before\n```tool_code\ndefault_api.read(path="a")\n```\nafter'));
expect(text).toContain("before");
expect(text).toContain("after");
expect(text).not.toContain("default_api");
});
it("yields the same calls when streamed character by character", () => {
const text = '```tool_code\ndefault_api.read(path="a.ts", count=7)\n```';
expect(parsedCalls("gemini", text, true)).toEqual([{ name: "read", arguments: { path: "a.ts", count: 7 } }]);
});
it("renders parallel calls as a list and round-trips through the scanner", () => {
const grammar = getInbandGrammar("gemini");
const rendered = grammar.renderAssistantToolCalls([
call("read", { path: "a" }),
call("write", { path: "b", content: "c" }),
]);
expect(rendered).toContain("```tool_code");
expect(rendered).toContain("[default_api.read(path=");
expect(rendered).toContain("default_api.write(path=");
expect(parsedCalls("gemini", rendered)).toEqual([
{ name: "read", arguments: { path: "a" } },
{ name: "write", arguments: { path: "b", content: "c" } },
]);
});
it("renders examples without a fence or print wrapper", () => {
const grammar = getInbandGrammar("gemini");
expect(grammar.renderToolCall(call("read", { path: "a.ts" }), { example: true })).toBe('read(path="a.ts")');
expect(grammar.renderAssistantToolCalls([call("read", { path: "a.ts" })], { example: true })).toBe(
'read(path="a.ts")',
);
});
it("escapes special characters on render and decodes them on parse", () => {
const grammar = getInbandGrammar("gemini");
const rendered = grammar.renderAssistantToolCalls([call("write", { content: 'a "b"\n\tc\\d' })]);
expect(parsedCalls("gemini", rendered)).toEqual([{ name: "write", arguments: { content: 'a "b"\n\tc\\d' } }]);
});
});
describe("gemma grammar (token-delimited call:NAME{…})", () => {
it("parses a single call with string and scalar args", () => {
const calls = parsedCalls("gemma", '<|tool_call>call:read{path:<|"|>a.ts<|"|>,count:2}<tool_call|>');
expect(calls).toEqual([{ name: "read", arguments: { path: "a.ts", count: 2 } }]);
});
it('keeps commas and quotes inside <|"|> string values', () => {
const calls = parsedCalls("gemma", '<|tool_call>call:f{loc:<|"|>San Francisco, CA "downtown"<|"|>}<tool_call|>');
expect(calls[0]!.arguments).toEqual({ loc: 'San Francisco, CA "downtown"' });
});
it("keeps close-token text inside string values", () => {
const calls = parsedCalls(
"gemma",
'<|tool_call>call:read{path:<|"|>literal <tool_call|> marker, ok<|"|>,count:2}<tool_call|>',
true,
);
expect(calls).toEqual([{ name: "read", arguments: { path: "literal <tool_call|> marker, ok", count: 2 } }]);
});
it("parses scalars, lists, and nested objects", () => {
const calls = parsedCalls(
"gemma",
'<|tool_call>call:f{b:true,z:null,n:3,arr:[<|"|>a<|"|>,<|"|>b<|"|>],obj:{k:<|"|>v<|"|>}}<tool_call|>',
);
expect(calls[0]!.arguments).toEqual({ b: true, z: null, n: 3, arr: ["a", "b"], obj: { k: "v" } });
});
it("parses consecutive blocks as parallel calls", () => {
const calls = parsedCalls(
"gemma",
'<|tool_call>call:read{path:<|"|>a<|"|>}<tool_call|><|tool_call>call:write{path:<|"|>b<|"|>}<tool_call|>',
);
expect(calls).toEqual([
{ name: "read", arguments: { path: "a" } },
{ name: "write", arguments: { path: "b" } },
]);
});
it("yields the same call when streamed character by character", () => {
const text = '<|tool_call>call:read{path:<|"|>a.ts<|"|>,count:7}<tool_call|>';
expect(parsedCalls("gemma", text, true)).toEqual([{ name: "read", arguments: { path: "a.ts", count: 7 } }]);
});
it("renders calls that round-trip through the scanner", () => {
const grammar = getInbandGrammar("gemma");
const rendered = grammar.renderAssistantToolCalls([
call("read", { path: "a" }),
call("write", { path: "b", content: "c" }),
]);
expect(rendered).toBe(
'<|tool_call>call:read{path:<|"|>a<|"|>}<tool_call|><|tool_call>call:write{path:<|"|>b<|"|>,content:<|"|>c<|"|>}<tool_call|>',
);
expect(parsedCalls("gemma", rendered)).toEqual([
{ name: "read", arguments: { path: "a" } },
{ name: "write", arguments: { path: "b", content: "c" } },
]);
});
});
@@ -1,4 +1,5 @@
import { describe, expect, it } from "bun:test";
import type { AssistantMessage, Model, ToolCall, Usage } from "@oh-my-pi/pi-ai";
import {
createHarmonyAuditEvent,
detectHarmonyLeak,
@@ -7,11 +8,9 @@ import {
isHarmonyLeakMitigationTarget,
recoverHarmonyToolCall,
signalListLabel,
} from "@oh-my-pi/pi-agent-core/harmony-leak";
import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai";
} from "@oh-my-pi/pi-ai/utils/harmony-leak";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import corpus from "./fixtures/harmony-leak-corpus.json" with { type: "json" };
import { createAssistantMessage } from "./helpers";
interface CorpusPositive {
id: string;
@@ -30,6 +29,33 @@ const negatives = corpus.negatives as CorpusNegative[];
const codexModel: Model = getBundledModel("openai-codex", "gpt-5.4");
const anthropicModel: Model = getBundledModel("anthropic", "claude-sonnet-4-5");
function createAssistantMessage(
content: AssistantMessage["content"],
stopReason: AssistantMessage["stopReason"] = "stop",
): AssistantMessage {
return {
role: "assistant",
content,
api: "mock",
provider: "mock",
model: "mock-model",
usage: createUsage(),
stopReason,
timestamp: Date.now(),
};
}
function createUsage(): Usage {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
function makeToolCallMessage(toolName: string, input: string | null, argJson: string | null): AssistantMessage {
const callArgs: Record<string, unknown> =
input !== null ? { input } : argJson !== null ? (JSON.parse(argJson) as Record<string, unknown>) : {};
+259
View File
@@ -0,0 +1,259 @@
import { describe, expect, it } from "bun:test";
import type { AssistantMessage, Context, ToolCall, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai";
import {
createInbandScanner,
encodeInbandToolHistory,
type GrammarToolResult,
getInbandGrammar,
type InbandScanEvent,
parseInbandToolMessage,
renderInbandToolPrompt,
type ToolCallSyntax,
} from "@oh-my-pi/pi-ai/grammar";
const TOOLS = [
{
name: "read",
description: "Read a file",
parameters: {
type: "object",
properties: { path: { type: "string" }, count: { type: "number" } },
required: ["path"],
},
},
{
name: "write",
description: "Write a file",
parameters: {
type: "object",
properties: { path: { type: "string" }, content: { type: "string" } },
required: ["path", "content"],
},
},
] as unknown as NonNullable<Context["tools"]>;
const SYNTAXES: readonly ToolCallSyntax[] = [
"glm",
"hermes",
"kimi",
"xml",
"anthropic",
"deepseek",
"harmony",
"pi",
"qwen3",
"gemini",
"gemma",
];
function usage(): Usage {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
function assistant(content: AssistantMessage["content"]): AssistantMessage {
return {
role: "assistant",
content,
api: "mock",
provider: "mock",
model: "mock-model",
usage: usage(),
stopReason: "toolUse",
timestamp: 0,
};
}
function result(toolCallId: string, toolName: string, text: string, isError = false): ToolResultMessage {
return { role: "toolResult", toolCallId, toolName, content: [{ type: "text", text }], isError, timestamp: 0 };
}
function feedText(syntax: ToolCallSyntax, text: string): InbandScanEvent[] {
const scanner = createInbandScanner(syntax, { tools: TOOLS, parseThinking: true });
const events: InbandScanEvent[] = [];
for (const char of text) events.push(...scanner.feed(char));
events.push(...scanner.flush());
return events;
}
function toolEnds(events: readonly InbandScanEvent[]): Extract<InbandScanEvent, { type: "toolEnd" }>[] {
return events.filter((event): event is Extract<InbandScanEvent, { type: "toolEnd" }> => event.type === "toolEnd");
}
function firstRawBlock(syntax: ToolCallSyntax, text: string): string | undefined {
return toolEnds(feedText(syntax, text))[0]?.rawBlock;
}
function expectRawBlock(syntax: ToolCallSyntax, text: string, expected: string): void {
expect(firstRawBlock(syntax, text), syntax).toBe(expected);
}
describe("in-band tool grammars", () => {
it("renders a tool prompt for every syntax", () => {
for (const syntax of SYNTAXES) {
const prompt = renderInbandToolPrompt(TOOLS, syntax);
expect(prompt).toContain("<tools>");
expect(prompt).toContain("</tools>");
expect(prompt).toContain('"name":"read"');
expect(prompt).toContain(getInbandGrammar(syntax).prompt.trim().split("\n", 1)[0]!);
}
});
it("each grammar renders calls that its scanner parses back", () => {
const call: ToolCall = {
type: "toolCall",
id: "functions.read:0",
name: "read",
arguments: { path: "src/a.ts", count: 2 },
};
for (const syntax of SYNTAXES) {
const grammar = getInbandGrammar(syntax);
const rendered = grammar.renderAssistantToolCalls([call], { tools: TOOLS });
const calls = toolEnds(feedText(syntax, rendered));
expect(calls, syntax).toHaveLength(1);
expect(calls[0]!.name).toBe("read");
expect(calls[0]!.arguments).toEqual({ path: "src/a.ts", count: 2 });
}
});
it("captures exact raw tool call blocks for debugging", () => {
expectRawBlock(
"glm",
"<tool_call>read\n<arg_key>path</arg_key>\n<arg_value>src/a.ts</arg_value>\n</tool_call>",
"<tool_call>read\n<arg_key>path</arg_key>\n<arg_value>src/a.ts</arg_value>\n</tool_call>",
);
expectRawBlock(
"kimi",
'<|tool_calls_section_begin|><|tool_call_begin|>functions.read:0<|tool_call_argument_begin|> {"path":"src/a.ts"}\n<|tool_call_end|><|tool_calls_section_end|>',
'<|tool_call_begin|>functions.read:0<|tool_call_argument_begin|> {"path":"src/a.ts"}\n<|tool_call_end|>',
);
expectRawBlock(
"deepseek",
'<|DSML|tool_calls>\n<|DSML|invoke name="read">\n <|DSML|parameter name="path" string="true">src/a.ts</|DSML|parameter>\n</|DSML|invoke>\n</|DSML|tool_calls>',
'<|DSML|invoke name="read">\n <|DSML|parameter name="path" string="true">src/a.ts</|DSML|parameter>\n</|DSML|invoke>',
);
expectRawBlock(
"xml",
'<function_calls>\n<invoke name="read"><parameter name="path" string="true">src/a.ts</parameter></invoke>\n</function_calls>',
'<invoke name="read"><parameter name="path" string="true">src/a.ts</parameter></invoke>',
);
expectRawBlock(
"harmony",
'<|start|>assistant<|channel|>commentary to=functions.read<|message|>{"path":"src/a.ts"}<|call|>',
'<|start|>assistant<|channel|>commentary to=functions.read<|message|>{"path":"src/a.ts"}<|call|>',
);
expectRawBlock(
"pi",
'<call:write path="out.ts">\nhello\n</call:write>',
'<call:write path="out.ts">\nhello\n</call:write>',
);
});
it("projects raw tool blocks onto parsed ToolCall content", () => {
const raw = '<|start|>assistant<|channel|>commentary to=functions.read<|message|>{"path":"src/a.ts"}<|call|>';
const parsed = parseInbandToolMessage(assistant([{ type: "text", text: raw }]), "harmony", TOOLS);
const call = parsed.content.find((block): block is ToolCall => block.type === "toolCall");
expect(call?.rawBlock).toBe(raw);
});
it("stops before hallucinated Anthropic function results", () => {
const parsed = parseInbandToolMessage(
assistant([
{
type: "text",
text: '<invoke name="read"><parameter name="path">rubygems.ts:85-93</parameter></invoke>\n<function_results>\n<result>\n<tool_name>read</tool_name>\n<stdout>[rubygems.ts#A1B2]</stdout>\n</result>\n</function_results>\n<invoke name="edit"><parameter name="input">[rubygems.ts#A1B2]\nSWAP 89..89:\n+ fake</parameter></invoke>',
},
]),
"anthropic",
TOOLS,
);
const calls = parsed.content.filter((block): block is ToolCall => block.type === "toolCall");
expect(calls.map(call => call.name)).toEqual(["read"]);
expect(calls[0]?.arguments).toEqual({ path: "rubygems.ts:85-93" });
});
it("keeps result rendering in the owning grammar", () => {
const resultBlock: GrammarToolResult = {
id: "functions.read:0",
name: "read",
index: 0,
text: "FILE",
isError: false,
};
expect(getInbandGrammar("glm").renderToolResults([resultBlock])).toBe(
"<observation>\n<tool_response>\nFILE\n</tool_response>\n</observation>",
);
expect(getInbandGrammar("deepseek").renderToolResults([resultBlock])).toBe(
"<|tool▁output▁begin|>FILE<|tool▁output▁end|>",
);
expect(getInbandGrammar("kimi").renderToolResults([resultBlock])).toBe(
"<|im_system|>read<|im_middle|>## Return of functions.read:0\nFILE<|im_end|>",
);
expect(getInbandGrammar("harmony").renderToolResults([resultBlock])).toBe(
"<|start|>functions.read to=assistant<|channel|>commentary<|message|>FILE<|end|>",
);
expect(getInbandGrammar("anthropic").renderToolResults([resultBlock])).toBe(
"<function_results>\n<result>\n<tool_name>read</tool_name>\n<stdout>FILE</stdout>\n</result>\n</function_results>",
);
expect(getInbandGrammar("qwen3").renderToolResults([resultBlock])).toBe(
"<tool_response>\nFILE\n</tool_response>",
);
expect(getInbandGrammar("pi").renderToolResults([resultBlock])).toBe("<tool_response>\nFILE\n</tool_response>");
expect(getInbandGrammar("gemini").renderToolResults([resultBlock])).toBe("```tool_outputs\nFILE\n```");
expect(getInbandGrammar("gemma").renderToolResults([resultBlock])).toBe(
'<|tool_response>response:read{output:<|"|>FILE<|"|>}<tool_response|>',
);
});
it("encodes assistant calls and tool results through the selected grammar", () => {
const history: Context["messages"] = [
{ role: "user", content: "hi", timestamp: 0 },
assistant([
{ type: "text", text: "let me read" },
{ type: "toolCall", id: "functions.read:0", name: "read", arguments: { path: "a.ts" } },
]),
result("functions.read:0", "read", "FILE A"),
];
const enc = encodeInbandToolHistory(history, "kimi", TOOLS);
expect(enc[0]).toBe(history[0]);
expect(enc[1]!.role).toBe("assistant");
expect(enc[2]!.role).toBe("user");
const assistantBlock = (enc[1] as AssistantMessage).content[0]!;
const assistantText = assistantBlock.type === "text" ? assistantBlock.text : "";
expect(assistantText).toContain("<|tool_calls_section_begin|>");
expect(assistantText).toContain("functions.read:0");
const resultText =
Array.isArray(enc[2]!.content) && enc[2]!.content[0]!.type === "text" ? enc[2]!.content[0]!.text : "";
expect(resultText).toBe("<|im_system|>read<|im_middle|>## Return of functions.read:0\nFILE A<|im_end|>");
});
it("streams string arguments incrementally for GLM", () => {
const text = getInbandGrammar("glm").renderAssistantToolCalls(
[
{
type: "toolCall",
id: "c1",
name: "write",
arguments: { path: "out.ts", content: "line1\nconst x = `a`;" },
},
],
{ tools: TOOLS },
);
const deltas = feedText("glm", text)
.filter(
(event): event is Extract<InbandScanEvent, { type: "toolArgDelta" }> =>
event.type === "toolArgDelta" && event.key === "content",
)
.map(event => event.delta)
.join("");
expect(deltas).toBe("line1\nconst x = `a`;");
});
});
-170
View File
@@ -1,170 +0,0 @@
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { streamBedrock } from "@oh-my-pi/pi-ai/providers/amazon-bedrock";
import { clearAwsCredentialCache } from "@oh-my-pi/pi-ai/providers/aws-credentials";
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
const model: Model<"bedrock-converse-stream"> = buildModel({
id: "zai.glm-5",
name: "GLM-5",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
baseUrl: "https://bedrock-runtime.us-west-2.amazonaws.com",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 131_072,
maxTokens: 16_384,
});
const context: Context = {
systemPrompt: [],
messages: [{ role: "user", content: "say hi", timestamp: Date.now() }],
};
const awsEnvKeys = [
"AWS_ACCESS_KEY_ID",
"AWS_SECRET_ACCESS_KEY",
"AWS_SESSION_TOKEN",
"AWS_PROFILE",
"AWS_REGION",
"AWS_DEFAULT_REGION",
"AWS_CONFIG_FILE",
"AWS_SHARED_CREDENTIALS_FILE",
"AWS_EC2_METADATA_DISABLED",
"AWS_BEARER_TOKEN_BEDROCK",
"AWS_BEDROCK_SKIP_AUTH",
] as const;
function snapshotAwsEnv(): () => void {
const previous = new Map<string, string | undefined>();
for (const key of awsEnvKeys) previous.set(key, process.env[key]);
return () => {
for (const key of awsEnvKeys) {
const value = previous.get(key);
if (value === undefined) delete process.env[key];
else process.env[key] = value;
}
clearAwsCredentialCache();
};
}
describe("issue #1399: Bedrock bearer token precedence", () => {
it("uses AWS_BEARER_TOKEN_BEDROCK without invoking profile credential_process", async () => {
const restoreAwsEnv = snapshotAwsEnv();
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-bedrock-auth-"));
try {
const configPath = path.join(tempDir, "config");
await Bun.write(
configPath,
[
"[default]",
"region = us-west-2",
"credential_process = /bin/sh -c 'echo should-not-run >&2; exit 17'",
"",
].join("\n"),
);
delete process.env.AWS_ACCESS_KEY_ID;
delete process.env.AWS_SECRET_ACCESS_KEY;
delete process.env.AWS_SESSION_TOKEN;
delete process.env.AWS_PROFILE;
delete process.env.AWS_DEFAULT_REGION;
delete process.env.AWS_BEDROCK_SKIP_AUTH;
process.env.AWS_REGION = "us-west-2";
process.env.AWS_CONFIG_FILE = configPath;
process.env.AWS_SHARED_CREDENTIALS_FILE = path.join(tempDir, "credentials");
process.env.AWS_EC2_METADATA_DISABLED = "true";
process.env.AWS_BEARER_TOKEN_BEDROCK = "bedrock-api-key";
clearAwsCredentialCache();
let requestHeaders: Headers | undefined;
const fetchMock: FetchImpl = async (_input, init) => {
requestHeaders = new Headers(init?.headers);
return new Response('{"message":"unauthorized"}', { status: 401 });
};
const result = await streamBedrock(model, context, { fetch: fetchMock }).result();
expect(requestHeaders?.get("authorization")).toBe("Bearer bedrock-api-key");
expect(requestHeaders?.has("x-amz-date")).toBe(false);
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toContain("Bedrock HTTP 401");
expect(result.errorMessage).not.toContain("credential_process");
} finally {
restoreAwsEnv();
await fs.rm(tempDir, { recursive: true, force: true });
}
});
it("ignores agent sentinel apiKey when AWS_BEARER_TOKEN_BEDROCK is available", async () => {
const restoreAwsEnv = snapshotAwsEnv();
try {
delete process.env.AWS_ACCESS_KEY_ID;
delete process.env.AWS_SECRET_ACCESS_KEY;
delete process.env.AWS_SESSION_TOKEN;
delete process.env.AWS_PROFILE;
delete process.env.AWS_CONFIG_FILE;
delete process.env.AWS_DEFAULT_REGION;
delete process.env.AWS_BEDROCK_SKIP_AUTH;
process.env.AWS_REGION = "us-west-2";
process.env.AWS_SHARED_CREDENTIALS_FILE = path.join(os.tmpdir(), "missing-aws-credentials");
process.env.AWS_EC2_METADATA_DISABLED = "true";
process.env.AWS_BEARER_TOKEN_BEDROCK = "bedrock-api-key";
clearAwsCredentialCache();
let requestHeaders: Headers | undefined;
const fetchMock: FetchImpl = async (_input, init) => {
requestHeaders = new Headers(init?.headers);
return new Response('{"message":"unauthorized"}', { status: 401 });
};
const result = await streamBedrock(model, context, { apiKey: "<authenticated>", fetch: fetchMock }).result();
expect(requestHeaders?.get("authorization")).toBe("Bearer bedrock-api-key");
expect(requestHeaders?.get("authorization")).not.toBe("Bearer <authenticated>");
expect(requestHeaders?.has("x-amz-date")).toBe(false);
expect(result.stopReason).toBe("error");
} finally {
restoreAwsEnv();
}
});
it("ignores agent sentinel apiKey when signing with AWS credentials", async () => {
const restoreAwsEnv = snapshotAwsEnv();
try {
process.env.AWS_ACCESS_KEY_ID = "AKIDEXAMPLE";
process.env.AWS_SECRET_ACCESS_KEY = "wJalrXUtnFEMI/K7MDENG+bPxRfiCYEXAMPLEKEY";
delete process.env.AWS_SESSION_TOKEN;
delete process.env.AWS_PROFILE;
delete process.env.AWS_CONFIG_FILE;
delete process.env.AWS_SHARED_CREDENTIALS_FILE;
delete process.env.AWS_DEFAULT_REGION;
delete process.env.AWS_BEARER_TOKEN_BEDROCK;
delete process.env.AWS_BEDROCK_SKIP_AUTH;
process.env.AWS_REGION = "us-west-2";
process.env.AWS_EC2_METADATA_DISABLED = "true";
clearAwsCredentialCache();
let requestHeaders: Headers | undefined;
const fetchMock: FetchImpl = async (_input, init) => {
requestHeaders = new Headers(init?.headers);
return new Response('{"message":"unauthorized"}', { status: 401 });
};
const result = await streamBedrock(model, context, { apiKey: "<authenticated>", fetch: fetchMock }).result();
const authorization = requestHeaders?.get("authorization");
expect(authorization).toStartWith("AWS4-HMAC-SHA256 ");
expect(authorization).toContain("Credential=AKIDEXAMPLE/");
expect(authorization).not.toBe("Bearer <authenticated>");
expect(requestHeaders?.has("x-amz-date")).toBe(true);
expect(result.stopReason).toBe("error");
} finally {
restoreAwsEnv();
}
});
});
+11 -11
View File
@@ -80,7 +80,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
const fetchMock = createMockFetch([
toolCallChunk(model, {
name: "edit",
arguments: { input: "[foo.ts#A1B2]\nreplace 91..91:\n+ " },
arguments: { input: "[foo.ts#A1B2]\nSWAP 91.=91:\n+ " },
}),
toolCallChunk(model, {
arguments: { input: 'const out = await executeTool("nuke", { path: "x" }, ctx);' },
@@ -100,7 +100,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
id: "call-minimax-1",
name: "edit",
arguments: {
input: '[foo.ts#A1B2]\nreplace 91..91:\n+ const out = await executeTool("nuke", { path: "x" }, ctx);',
input: '[foo.ts#A1B2]\nSWAP 91.=91:\n+ const out = await executeTool("nuke", { path: "x" }, ctx);',
},
},
]);
@@ -115,10 +115,10 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
const fetchMock = createMockFetch([
toolCallChunk(model, {
name: "edit",
arguments: { input: "[foo.ts#A1B2]\nreplace 91..91:" },
arguments: { input: "[foo.ts#A1B2]\nSWAP 91.=91:" },
}),
toolCallChunk(model, {
arguments: { input: "[foo.ts#A1B2]\nreplace 91..91:\n+new" },
arguments: { input: "[foo.ts#A1B2]\nSWAP 91.=91:\n+new" },
}),
stopChunk(model),
"[DONE]",
@@ -134,7 +134,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
type: "toolCall",
id: "call-minimax-1",
name: "edit",
arguments: { input: "[foo.ts#A1B2]\nreplace 91..91:\n+new" },
arguments: { input: "[foo.ts#A1B2]\nSWAP 91.=91:\n+new" },
},
]);
});
@@ -142,7 +142,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
it("preserves keys that only appear in earlier chunks instead of dropping them with later chunks", async () => {
const model = getBundledModel<"openai-completions">("minimax-code-cn", "MiniMax-M3");
const fetchMock = createMockFetch([
toolCallChunk(model, { name: "edit", arguments: { input: "[foo.ts#A1B2]\ndelete 5" } }),
toolCallChunk(model, { name: "edit", arguments: { input: "[foo.ts#A1B2]\nDEL 5" } }),
toolCallChunk(model, { arguments: { dryRun: true } }),
stopChunk(model),
"[DONE]",
@@ -158,7 +158,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
type: "toolCall",
id: "call-minimax-1",
name: "edit",
arguments: { input: "[foo.ts#A1B2]\ndelete 5", dryRun: true },
arguments: { input: "[foo.ts#A1B2]\nDEL 5", dryRun: true },
},
]);
});
@@ -175,7 +175,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
const fetchMock = createMockFetch([
toolCallChunk(model, {
name: "edit",
arguments: { input: "[foo.ts#A1B2]\nreplace 91..91:\n+ " },
arguments: { input: "[foo.ts#A1B2]\nSWAP 91.=91:\n+ " },
}),
toolCallChunk(model, {
arguments: { input: 'const out = await executeTool("nuke", { path: "x" }, ctx);' },
@@ -193,7 +193,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
}
const expected = {
input: '[foo.ts#A1B2]\nreplace 91..91:\n+ const out = await executeTool("nuke", { path: "x" }, ctx);',
input: '[foo.ts#A1B2]\nSWAP 91.=91:\n+ const out = await executeTool("nuke", { path: "x" }, ctx);',
};
// Source-side merged result (what `block.arguments` is set to in `finishToolCallBlock`).
expect(toolCallEndArgs).toEqual(expected);
@@ -208,7 +208,7 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
// the proxy still concatenates ("" then the final delta) and parses to the same args.
const model = getBundledModel<"openai-completions">("minimax-code-cn", "MiniMax-M3");
const fetchMock = createMockFetch([
toolCallChunk(model, { name: "edit", arguments: { input: "[foo.ts#A1B2]\ndelete 5" } }),
toolCallChunk(model, { name: "edit", arguments: { input: "[foo.ts#A1B2]\nDEL 5" } }),
stopChunk(model),
"[DONE]",
]);
@@ -218,6 +218,6 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
for await (const event of s) {
if (event.type === "toolcall_delta") accumulated += event.delta;
}
expect(JSON.parse(accumulated)).toEqual({ input: "[foo.ts#A1B2]\ndelete 5" });
expect(JSON.parse(accumulated)).toEqual({ input: "[foo.ts#A1B2]\nDEL 5" });
});
});
@@ -0,0 +1,94 @@
import { describe, expect, it } from "bun:test";
import { jsonSchemaToTypeScript, toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema";
import { z } from "zod/v4";
describe("jsonSchemaToTypeScript", () => {
it("renders objects with optional markers and JSDoc descriptions", () => {
const ts = jsonSchemaToTypeScript({
type: "object",
properties: {
query: { type: "string", description: "search query" },
limit: { type: "number", description: "max results" },
},
required: ["query"],
});
expect(ts).toContain("/** search query */");
expect(ts).toContain("query: string;");
expect(ts).toContain("/** max results */");
expect(ts).toContain("limit?: number;");
});
it("renders enums and consts as literal unions", () => {
const ts = jsonSchemaToTypeScript({
type: "object",
properties: {
recency: { type: "string", enum: ["day", "week", "month"] },
kind: { const: "fixed" },
},
required: ["recency", "kind"],
});
expect(ts).toContain('recency: "day" | "week" | "month";');
expect(ts).toContain('kind: "fixed";');
});
it("renders arrays, tuples, and records", () => {
const ts = jsonSchemaToTypeScript({
type: "object",
properties: {
tags: { type: "array", items: { type: "string" } },
pair: { type: "array", prefixItems: [{ type: "string" }, { type: "number" }] },
meta: { type: "object", additionalProperties: { type: "number" } },
rows: {
type: "array",
items: { type: "object", properties: { id: { type: "string" } }, required: ["id"] },
},
},
required: ["tags", "pair", "meta", "rows"],
});
expect(ts).toContain("tags: string[];");
expect(ts).toContain("pair: [string, number];");
expect(ts).toContain("meta: Record<string, number>;");
// Object-valued array elements expand to Array<{ … }> rather than inline `[]`.
expect(ts).toContain("rows: Array<{");
expect(ts).toContain("id: string;");
});
it("renders nullable unions from both type-arrays and anyOf", () => {
const ts = jsonSchemaToTypeScript({
type: "object",
properties: {
a: { type: ["string", "null"] },
b: { anyOf: [{ type: "number" }, { type: "null" }] },
},
required: ["a", "b"],
});
expect(ts).toContain("a: string | null;");
expect(ts).toContain("b: number | null;");
});
it("resolves a local $ref against $defs", () => {
const ts = jsonSchemaToTypeScript({
type: "object",
properties: { node: { $ref: "#/$defs/Node" } },
required: ["node"],
$defs: { Node: { type: "object", properties: { value: { type: "number" } }, required: ["value"] } },
});
expect(ts).toContain("node: {");
expect(ts).toContain("value: number;");
});
it("renders an empty object schema as {}", () => {
expect(jsonSchemaToTypeScript({ type: "object", properties: {}, additionalProperties: false })).toBe("{}");
});
it("converts a Zod schema through the wire pipeline", () => {
const parameters = z.object({
name: z.string().describe("the name"),
count: z.number().int().optional(),
});
const ts = jsonSchemaToTypeScript(toolWireSchema({ name: "t", description: "", parameters }));
expect(ts).toContain("/** the name */");
expect(ts).toContain("name: string;");
expect(ts).toContain("count?: number;");
});
});
@@ -410,6 +410,300 @@ describe("openai-codex streaming", () => {
expect("lastParseLen" in toolCall).toBe(false);
});
it("routes interleaved function-call argument deltas to the matching open item", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
const context = createCodexTestContext();
// Two function calls are opened concurrently and the server interleaves
// `function_call_arguments.delta` events by `item_id`. With the old
// singleton current-block, every delta went to whichever item was added
// most recently; the `task` call ended up with `arguments = {}` and the
// sibling received the `task` payload (issue #2619). Each call must
// retain its own arguments and emit `toolcall_*` events against its own
// content index.
const taskArgs = '{"ops":[{"op":"start","task":"X"}]}';
const otherArgs = '{"input":"hello"}';
const events: Array<Record<string, unknown>> = [
{
type: "response.output_item.added",
output_index: 0,
item: { type: "function_call", id: "fc_task", call_id: "call_task", name: "task", arguments: "" },
},
{
type: "response.output_item.added",
output_index: 1,
item: { type: "function_call", id: "fc_other", call_id: "call_other", name: "other", arguments: "" },
},
{
type: "response.function_call_arguments.delta",
item_id: "fc_task",
output_index: 0,
delta: taskArgs.slice(0, 12),
},
{
type: "response.function_call_arguments.delta",
item_id: "fc_other",
output_index: 1,
delta: otherArgs.slice(0, 10),
},
{
type: "response.function_call_arguments.delta",
item_id: "fc_task",
output_index: 0,
delta: taskArgs.slice(12),
},
{
type: "response.function_call_arguments.delta",
item_id: "fc_other",
output_index: 1,
delta: otherArgs.slice(10),
},
// Stale delta for fc_task arriving after fc_other finishes must be dropped,
// not appended to fc_other.
{
type: "response.output_item.done",
output_index: 1,
item: { type: "function_call", id: "fc_other", call_id: "call_other", name: "other", arguments: otherArgs },
},
{
type: "response.function_call_arguments.delta",
item_id: "fc_other",
output_index: 1,
delta: "STALE",
},
{
type: "response.output_item.done",
output_index: 0,
item: { type: "function_call", id: "fc_task", call_id: "call_task", name: "task", arguments: taskArgs },
},
{
type: "response.completed",
response: {
id: "resp_1",
status: "completed",
usage: {
input_tokens: 5,
output_tokens: 3,
total_tokens: 8,
input_tokens_details: { cached_tokens: 0 },
},
},
},
];
const sse = `${events.map(e => `data: ${JSON.stringify(e)}`).join("\n\n")}\n\n`;
const fetchMock: FetchImpl = (async () =>
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } })) as FetchImpl;
const toolcallEnds: Array<{ contentIndex: number; name: string; argumentsJson: string }> = [];
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
const aem = streamOpenAICodexResponses(model, context, { apiKey: token, fetch: fetchMock });
(async () => {
for await (const event of aem) {
if (event.type !== "toolcall_end") continue;
toolcallEnds.push({
contentIndex: event.contentIndex,
name: event.toolCall.name,
argumentsJson: JSON.stringify(event.toolCall.arguments),
});
}
})();
const result = await aem.result();
const calls = result.content.filter(c => c.type === "toolCall");
expect(calls).toHaveLength(2);
const byName = new Map(calls.map(c => [c.name, c] as const));
expect(byName.get("task")?.arguments).toEqual({ ops: [{ op: "start", task: "X" }] });
expect(byName.get("other")?.arguments).toEqual({ input: "hello" });
// `task` is the FIRST opened block (index 0); a stale delta after fc_other
// closed must NOT have appended "STALE" anywhere.
expect(JSON.stringify(result.content)).not.toContain("STALE");
// Stream events must address each tool call by its own content index.
expect(toolcallEnds.find(e => e.name === "task")?.contentIndex).toBe(0);
expect(toolcallEnds.find(e => e.name === "other")?.contentIndex).toBe(1);
});
it("uses output_index to finalize idless function and custom tool calls", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
const context = createCodexTestContext();
const taskArgs = '{"tasks":[{"assignment":"fix it"}]}';
const patchInput = "*** Begin Patch\n*** End Patch";
const events: Array<Record<string, unknown>> = [
{
type: "response.output_item.added",
output_index: 0,
item: { type: "function_call", call_id: "call_task_no_id", name: "task", arguments: "" },
},
{
type: "response.output_item.added",
output_index: 1,
item: { type: "custom_tool_call", call_id: "call_patch_no_id", name: "apply_patch", input: "" },
},
{
type: "response.output_item.done",
output_index: 1,
item: { type: "custom_tool_call", call_id: "call_patch_no_id", name: "apply_patch", input: patchInput },
},
{
type: "response.output_item.done",
output_index: 0,
item: { type: "function_call", call_id: "call_task_no_id", name: "task", arguments: taskArgs },
},
{
type: "response.completed",
response: {
id: "resp_1",
status: "completed",
usage: {
input_tokens: 5,
output_tokens: 3,
total_tokens: 8,
input_tokens_details: { cached_tokens: 0 },
},
},
},
];
const sse = `${events.map(e => `data: ${JSON.stringify(e)}`).join("\n\n")}\n\n`;
const fetchMock: FetchImpl = (async () =>
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } })) as FetchImpl;
const toolcallEnds: Array<{ contentIndex: number; name: string }> = [];
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
const aem = streamOpenAICodexResponses(model, context, { apiKey: token, fetch: fetchMock });
(async () => {
for await (const event of aem) {
if (event.type !== "toolcall_end") continue;
toolcallEnds.push({ contentIndex: event.contentIndex, name: event.toolCall.name });
}
})();
const result = await aem.result();
const calls = result.content.filter(c => c.type === "toolCall");
const byName = new Map(calls.map(c => [c.name, c] as const));
expect(byName.get("task")?.arguments).toEqual({ tasks: [{ assignment: "fix it" }] });
expect(byName.get("apply_patch")?.arguments).toEqual({ input: patchInput });
expect(toolcallEnds.find(e => e.name === "task")?.contentIndex).toBe(0);
expect(toolcallEnds.find(e => e.name === "apply_patch")?.contentIndex).toBe(1);
});
it("routes fully keyless deltas/done to the latest open item via currentEntry", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
const context = createCodexTestContext();
// Pathological legacy/proxy stream: `output_item.added` carries no `id`
// AND no `output_index`, so neither keyed map ever receives the item.
// `function_call_arguments.delta` / `output_item.done` likewise lack
// both keys. The runtime must still route them via `currentEntry`
// (the latest live `output_item.added`) instead of dropping.
const taskArgs = '{"tasks":[{"assignment":"keyless"}]}';
const events: Array<Record<string, unknown>> = [
{
type: "response.output_item.added",
item: { type: "function_call", call_id: "call_keyless", name: "task", arguments: "" },
},
{ type: "response.function_call_arguments.delta", delta: taskArgs.slice(0, 12) },
{ type: "response.function_call_arguments.delta", delta: taskArgs.slice(12) },
{
type: "response.output_item.done",
item: { type: "function_call", call_id: "call_keyless", name: "task", arguments: taskArgs },
},
{
type: "response.completed",
response: {
id: "resp_keyless",
status: "completed",
usage: {
input_tokens: 1,
output_tokens: 1,
total_tokens: 2,
input_tokens_details: { cached_tokens: 0 },
},
},
},
];
const sse = `${events.map(e => `data: ${JSON.stringify(e)}`).join("\n\n")}\n\n`;
const fetchMock: FetchImpl = (async () =>
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } })) as FetchImpl;
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
const result = await streamOpenAICodexResponses(model, context, {
apiKey: token,
fetch: fetchMock as FetchImpl,
}).result();
const call = result.content.find(c => c.type === "toolCall");
expect(call?.name).toBe("task");
expect(call?.arguments).toEqual({ tasks: [{ assignment: "keyless" }] });
});
it("prefers a later id-only current item over an older output_index entry on unkeyed events", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
const context = createCodexTestContext();
// Mixed key shapes: the first call is output_index-keyed only, the
// second is id-only and is now the latest open item. An unkeyed delta
// must address the second call (currentEntry), not whatever the
// keyed-map iteration happens to surface first.
const idOnlyArgs = '{"input":"id-only-current"}';
const events: Array<Record<string, unknown>> = [
{
type: "response.output_item.added",
output_index: 0,
item: { type: "function_call", call_id: "call_old", name: "older", arguments: "" },
},
{
type: "response.output_item.added",
item: { type: "function_call", id: "fc_id_only", call_id: "call_new", name: "newer", arguments: "" },
},
// Keyless delta + done for the newer call — must route to fc_id_only.
{ type: "response.function_call_arguments.delta", delta: idOnlyArgs },
{
type: "response.output_item.done",
item: {
type: "function_call",
id: "fc_id_only",
call_id: "call_new",
name: "newer",
arguments: idOnlyArgs,
},
},
// Close the older one explicitly with its key so the test verifies isolation.
{
type: "response.output_item.done",
output_index: 0,
item: { type: "function_call", call_id: "call_old", name: "older", arguments: "{}" },
},
{
type: "response.completed",
response: {
id: "resp_mixed",
status: "completed",
usage: {
input_tokens: 1,
output_tokens: 1,
total_tokens: 2,
input_tokens_details: { cached_tokens: 0 },
},
},
},
];
const sse = `${events.map(e => `data: ${JSON.stringify(e)}`).join("\n\n")}\n\n`;
const fetchMock: FetchImpl = (async () =>
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } })) as FetchImpl;
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
const result = await streamOpenAICodexResponses(model, context, {
apiKey: token,
fetch: fetchMock as FetchImpl,
}).result();
const calls = result.content.filter(c => c.type === "toolCall");
const byName = new Map(calls.map(c => [c.name, c] as const));
expect(byName.get("newer")?.arguments).toEqual({ input: "id-only-current" });
expect(byName.get("older")?.arguments).toEqual({});
});
it("waits for caller abort when SSE streams only no-progress status events", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
@@ -0,0 +1,97 @@
import { describe, expect, it } from "bun:test";
import { wrapInbandToolStream } from "../src/grammar/owned-stream";
import type { AssistantMessage, ToolCall, Usage } from "../src/types";
import { AssistantMessageEventStream } from "../src/utils/event-stream";
const TOOLS = [
{
name: "echo",
description: "Echo a message.",
parameters: {
type: "object",
properties: { msg: { type: "string" } },
required: ["msg"],
},
},
];
const TOOL_CALL_TEXT = "<tool_call>echo\n<arg_key>msg</arg_key>\n<arg_value>hi</arg_value>\n</tool_call>\n";
const FABRICATION_TEXT = "<tool_response>\nFAKE RESULT\n</tool_response>";
function makeAssistant(content: AssistantMessage["content"]): AssistantMessage {
const usage: Usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
return {
role: "assistant",
content,
api: "mock",
provider: "mock",
model: "mock-model",
usage,
stopReason: "stop",
timestamp: 0,
};
}
// An assistant turn that issues one real tool call, then hallucinates its own
// tool result (the fabrication boundary).
function makeInner(): AssistantMessageEventStream {
const inner = new AssistantMessageEventStream();
const seed = makeAssistant([]);
inner.push({ type: "start", partial: seed });
inner.push({ type: "text_delta", contentIndex: 0, delta: TOOL_CALL_TEXT, partial: seed });
inner.push({ type: "text_delta", contentIndex: 0, delta: FABRICATION_TEXT, partial: seed });
const full = makeAssistant([{ type: "text", text: TOOL_CALL_TEXT + FABRICATION_TEXT }]);
inner.push({ type: "done", reason: "stop", message: full });
inner.end(full);
return inner;
}
describe("wrapInbandToolStream fabrication handling", () => {
it("aborts the provider on fabrication when abortOnFabrication is true (default)", async () => {
let aborted = false;
const wrapped = wrapInbandToolStream(makeInner(), TOOLS, "glm", () => {
aborted = true;
});
const message = await wrapped.result();
expect(aborted).toBe(true);
const calls = message.content.filter((b): b is ToolCall => b.type === "toolCall");
expect(calls).toHaveLength(1);
expect(calls[0]!.name).toBe("echo");
expect(calls[0]!.arguments).toEqual({ msg: "hi" });
// The fabricated continuation is dropped, not surfaced as text.
const text = message.content.map(b => (b.type === "text" ? b.text : "")).join("");
expect(text).not.toContain("FAKE RESULT");
});
it("keeps the provider running and discards the continuation when abortOnFabrication is false", async () => {
let aborted = false;
const wrapped = wrapInbandToolStream(
makeInner(),
TOOLS,
"glm",
() => {
aborted = true;
},
false,
);
const message = await wrapped.result();
// No premature abort — the request is allowed to finish.
expect(aborted).toBe(false);
// Same canonical outcome: the real call is kept, the fabrication discarded.
const calls = message.content.filter((b): b is ToolCall => b.type === "toolCall");
expect(calls).toHaveLength(1);
expect(calls[0]!.name).toBe("echo");
expect(calls[0]!.arguments).toEqual({ msg: "hi" });
const text = message.content.map(b => (b.type === "text" ? b.text : "")).join("");
expect(text).not.toContain("FAKE RESULT");
});
});
@@ -0,0 +1,155 @@
import { describe, expect, it } from "bun:test";
import { wrapInbandToolStream } from "../src/grammar/owned-stream";
import type { AssistantMessage, AssistantMessageEvent, ThinkingContent, ToolCall, Usage } from "../src/types";
import { AssistantMessageEventStream } from "../src/utils/event-stream";
const TOOLS = [
{
name: "todo",
description: "Manage the todo list.",
parameters: {
type: "object",
properties: { ops: { type: "array" } },
required: ["ops"],
},
},
];
function makeAssistant(content: AssistantMessage["content"]): AssistantMessage {
const usage: Usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
return {
role: "assistant",
content,
api: "mock",
provider: "mock",
model: "mock-model",
usage,
stopReason: "toolUse",
timestamp: 0,
};
}
// Drive an inner provider stream the way openai-completions does: a single
// growing `output` message whose `content` each event's `partial` points at.
function drive(
build: (push: (event: AssistantMessageEvent) => void, out: AssistantMessage) => void,
): AssistantMessageEventStream {
const inner = new AssistantMessageEventStream();
const out = makeAssistant([]);
inner.push({ type: "start", partial: out });
build(event => inner.push(event), out);
inner.push({ type: "done", reason: out.stopReason === "length" ? "length" : "toolUse", message: out });
inner.end(out);
return inner;
}
// Gemini (via OpenRouter) keeps emitting native `tool_calls` even in owned mode
// where no `tools` are sent — the in-band scanner only reconstructs calls from
// `tool_code` text, so the projector must forward native calls (streamed) rather
// than dropping them.
function geminiNativeOnly(): AssistantMessageEventStream {
return drive((push, out) => {
const thinking: ThinkingContent = { type: "thinking", thinking: "Checking the todo list." };
out.content.push(thinking);
push({ type: "thinking_start", contentIndex: 0, partial: out });
push({ type: "thinking_delta", contentIndex: 0, delta: thinking.thinking, partial: out });
push({ type: "thinking_end", contentIndex: 0, content: thinking.thinking, partial: out });
const block: ToolCall = { type: "toolCall", id: "tool_todo_abc", name: "todo", arguments: {} };
out.content.push(block);
push({ type: "toolcall_start", contentIndex: 1, partial: out });
push({ type: "toolcall_delta", contentIndex: 1, delta: '{"ops":[{"op":"view"}]}', partial: out });
block.arguments = { ops: [{ op: "view" }] };
push({ type: "toolcall_end", contentIndex: 1, toolCall: block, partial: out });
});
}
// A nameless native "ghost" part (Gemini emits these beside a real call) must be
// dropped, while the real native call is still forwarded.
function ghostThenRealNative(): AssistantMessageEventStream {
return drive((push, out) => {
const ghost: ToolCall = { type: "toolCall", id: "", name: "", arguments: {} };
out.content.push(ghost);
push({ type: "toolcall_start", contentIndex: 0, partial: out });
push({ type: "toolcall_end", contentIndex: 0, toolCall: ghost, partial: out });
const real: ToolCall = {
type: "toolCall",
id: "tool_todo_real",
name: "todo",
arguments: { ops: [{ op: "view" }] },
};
out.content.push(real);
push({ type: "toolcall_start", contentIndex: 1, partial: out });
push({ type: "toolcall_end", contentIndex: 1, toolCall: real, partial: out });
});
}
// The duplicate-call report: Gemini writes a real in-band `tool_code` call AND
// also emits a native `functionCall`. Exactly one call must survive — the
// channel lock dedupes structurally, never by guessing from emptiness.
function inbandPlusNative(): AssistantMessageEventStream {
return drive((push, out) => {
const text = 'Sure.\n```tool_code\ndefault_api.todo(ops=[{"op": "view"}])\n```\n';
const textBlock = { type: "text" as const, text };
out.content.push(textBlock);
push({ type: "text_delta", contentIndex: 0, delta: text, partial: out });
const nativeDup: ToolCall = {
type: "toolCall",
id: "tool_todo_native",
name: "todo",
arguments: { ops: [{ op: "view" }] },
};
out.content.push(nativeDup);
push({ type: "toolcall_start", contentIndex: 1, partial: out });
push({ type: "toolcall_end", contentIndex: 1, toolCall: nativeDup, partial: out });
});
}
async function collect(stream: AssistantMessageEventStream): Promise<{ message: AssistantMessage; events: string[] }> {
const events: string[] = [];
for await (const event of stream) events.push(event.type);
return { message: await stream.result(), events };
}
describe("wrapInbandToolStream native tool-call passthrough", () => {
it("streams a provider-native tool call that arrives without in-band text", async () => {
const { message, events } = await collect(wrapInbandToolStream(geminiNativeOnly(), TOOLS, "gemini"));
const calls = message.content.filter((b): b is ToolCall => b.type === "toolCall");
expect(calls).toHaveLength(1);
expect(calls[0]!.name).toBe("todo");
expect(calls[0]!.id).toBe("tool_todo_abc");
expect(calls[0]!.arguments).toEqual({ ops: [{ op: "view" }] });
// Reasoning is preserved alongside the forwarded call.
expect(message.content.some(b => b.type === "thinking")).toBe(true);
// A turn with a tool call is "toolUse", never a content-less "stop".
expect(message.stopReason).toBe("toolUse");
// The full lifecycle streams live (not materialized in one shot at the end).
expect(events).toContain("toolcall_start");
expect(events).toContain("toolcall_delta");
expect(events).toContain("toolcall_end");
});
it("drops a nameless native ghost but keeps the real native call", async () => {
const { message } = await collect(wrapInbandToolStream(ghostThenRealNative(), TOOLS, "gemini"));
const calls = message.content.filter((b): b is ToolCall => b.type === "toolCall");
expect(calls).toHaveLength(1);
expect(calls[0]!.name).toBe("todo");
expect(calls[0]!.id).toBe("tool_todo_real");
});
it("emits exactly one call when the model uses both the in-band and native channels", async () => {
const { message } = await collect(wrapInbandToolStream(inbandPlusNative(), TOOLS, "gemini"));
const calls = message.content.filter((b): b is ToolCall => b.type === "toolCall");
// No double-dispatch, regardless of which channel won the lock.
expect(calls).toHaveLength(1);
expect(calls[0]!.name).toBe("todo");
expect(calls[0]!.arguments).toEqual({ ops: [{ op: "view" }] });
});
});
@@ -42,6 +42,163 @@ describe("Tool argument coercion", () => {
expect(typeof result.label).toBe("string");
});
it("stringifies object values when schema expects string", () => {
const tool: Tool = {
name: "object-string",
description: "",
parameters: z.object({ payload: z.string() }),
};
const result = validateToolArguments(tool, {
type: "toolCall",
id: "call-object-string",
name: "object-string",
arguments: { payload: { a: 1, nested: ["x"] } },
}) as { payload: string };
expect(result.payload).toBe('{"a":1,"nested":["x"]}');
});
it("stringifies array values when schema expects string", () => {
const tool: Tool = {
name: "array-string",
description: "",
parameters: z.object({ payload: z.string() }),
};
const result = validateToolArguments(tool, {
type: "toolCall",
id: "call-array-string",
name: "array-string",
arguments: { payload: ["a", 2, true] },
}) as { payload: string };
expect(result.payload).toBe('["a",2,true]');
});
it("coerces numeric 0 and 1 to booleans", () => {
const tool: Tool = {
name: "numeric-booleans",
description: "",
parameters: z.object({ enabled: z.boolean(), disabled: z.boolean() }),
};
const result = validateToolArguments(tool, {
type: "toolCall",
id: "call-numeric-booleans",
name: "numeric-booleans",
arguments: { enabled: 1, disabled: 0 },
}) as { enabled: boolean; disabled: boolean };
expect(result).toEqual({ enabled: true, disabled: false });
});
it("coerces booleans to numeric 0 and 1", () => {
const tool: Tool = {
name: "boolean-numbers",
description: "",
parameters: z.object({ enabled: z.number(), disabled: z.number().int() }),
};
const result = validateToolArguments(tool, {
type: "toolCall",
id: "call-boolean-numbers",
name: "boolean-numbers",
arguments: { enabled: true, disabled: false },
}) as { enabled: number; disabled: number };
expect(result).toEqual({ enabled: 1, disabled: 0 });
});
it("rejects numeric boolean values other than 0 or 1", () => {
const tool: Tool = {
name: "invalid-numeric-boolean",
description: "",
parameters: z.object({ enabled: z.boolean() }),
};
expect(() =>
validateToolArguments(tool, {
type: "toolCall",
id: "call-invalid-numeric-boolean",
name: "invalid-numeric-boolean",
arguments: { enabled: 2 },
}),
).toThrow('Validation failed for tool "invalid-numeric-boolean"');
});
it("keeps raw in-band blocks out of validation errors", () => {
const tool: Tool = {
name: "raw-debug",
description: "",
parameters: z.object({ input: z.string() }),
};
const rawBlock = '<|start|>assistant<|channel|>commentary to=functions.edit <|message|>{"input":"x"}}<|call|>';
expect(() =>
validateToolArguments(tool, {
type: "toolCall",
id: "call-raw-debug",
name: "raw-debug",
arguments: {},
rawBlock,
}),
).toThrow('Validation failed for tool "raw-debug"');
expect(() =>
validateToolArguments(tool, {
type: "toolCall",
id: "call-raw-debug",
name: "raw-debug",
arguments: {},
rawBlock,
}),
).not.toThrow(rawBlock);
});
it("coerces common string boolean forms", () => {
const tool: Tool = {
name: "string-booleans",
description: "",
parameters: z.object({
t: z.boolean(),
f: z.boolean(),
one: z.boolean(),
zero: z.boolean(),
yes: z.boolean(),
no: z.boolean(),
on: z.boolean(),
off: z.boolean(),
}),
};
const result = validateToolArguments(tool, {
type: "toolCall",
id: "call-string-booleans",
name: "string-booleans",
arguments: {
t: "TRUE",
f: "false",
one: "1",
zero: "0",
yes: "yes",
no: "NO",
on: "on",
off: "OFF",
},
}) as Record<string, boolean>;
expect(result).toEqual({
t: true,
f: false,
one: true,
zero: false,
yes: true,
no: false,
on: true,
off: false,
});
});
it("parses JSON arrays in string values when schema expects array", () => {
const tool: Tool = {
name: "t3",
+192
View File
@@ -0,0 +1,192 @@
import { describe, expect, it } from "bun:test";
import { renderToolExamples } from "../src/grammar/examples";
import type { InbandTool } from "../src/grammar/types";
describe("renderToolExamples", () => {
it("renders call example in anthropic format", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: {
type: "object",
properties: {
paths: { type: "array", items: { type: "string" } },
},
required: ["paths"],
},
examples: [
{
caption: "Find files",
call: { paths: ["src/**/*.ts"] },
},
],
};
const rendered = renderToolExamples(tool, "anthropic");
expect(rendered).toContain("<examples>");
expect(rendered).toContain("# Find files");
expect(rendered).toContain('<invoke name="find">');
expect(rendered).toContain('<parameter name="paths"');
expect(rendered).toContain("</examples>");
});
it("renders call example in pi format", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: {
type: "object",
properties: {
paths: { type: "array", items: { type: "string" } },
},
required: ["paths"],
},
examples: [
{
caption: "Find files",
call: { paths: ["src/**/*.ts"] },
},
],
};
const rendered = renderToolExamples(tool, "pi");
expect(rendered).toContain("<call:find>");
expect(rendered).toContain("<paths>");
expect(rendered).toContain("src/**/*.ts");
});
it("renders call example in hermes format", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: {
type: "object",
properties: {
paths: { type: "array", items: { type: "string" } },
},
required: ["paths"],
},
examples: [
{
caption: "Find files",
call: { paths: ["src/**/*.ts"] },
},
],
};
const rendered = renderToolExamples(tool, "hermes");
expect(rendered).toContain("<tool_call>");
expect(rendered).toContain('"name":"find"');
expect(rendered).toContain('"paths"');
});
it("renders harmony call example as bare JSON without the message envelope", () => {
const tool: InbandTool = {
name: "irc",
description: "IRC.",
parameters: {
type: "object",
properties: {
op: { type: "string" },
to: { type: "string" },
message: { type: "string" },
},
required: ["op"],
},
examples: [
{
caption: "Broadcast",
call: { op: "send", to: "all", message: "hi" },
},
],
};
const rendered = renderToolExamples(tool, "harmony");
expect(rendered).toContain('{"op":"send","to":"all","message":"hi"}');
// The verbose harmony envelope must be stripped inside <example> blocks.
expect(rendered).not.toContain("<|start|>");
expect(rendered).not.toContain("<|channel|>");
expect(rendered).not.toContain("<|message|>");
expect(rendered).not.toContain("<|call|>");
});
it("returns empty string for empty examples", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: { type: "object", properties: {} },
examples: [],
};
expect(renderToolExamples(tool, "anthropic")).toBe("");
});
it("renders compare examples with WRONG and RIGHT", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: {
type: "object",
properties: {
paths: { type: "array", items: { type: "string" } },
},
required: ["paths"],
},
examples: [
{
caption: "Avoid broad scans",
bad: { paths: ["**/*.ts"] },
good: { paths: ["src/**/*.ts"] },
},
],
};
const rendered = renderToolExamples(tool, "anthropic");
expect(rendered).toContain("WRONG:");
expect(rendered).toContain("RIGHT:");
expect(rendered).toContain('<parameter name="paths"');
expect(rendered).toContain('["**/*.ts"]');
});
it("injects the intent-field placeholder when intentField is provided", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: {
type: "object",
properties: {
_i: { type: "string" },
paths: { type: "array", items: { type: "string" } },
},
required: ["_i", "paths"],
},
examples: [
{
caption: "Find files",
call: { paths: ["src/**/*.ts"] },
},
],
};
const rendered = renderToolExamples(tool, "anthropic", "_i");
expect(rendered).toContain('<parameter name="_i"');
expect(rendered).toContain("…");
// Placeholder leads the args, matching schema-injection order.
expect(rendered.indexOf('name="_i"')).toBeLessThan(rendered.indexOf('name="paths"'));
});
it("omits the intent-field placeholder when intentField is undefined", () => {
const tool: InbandTool = {
name: "find",
description: "Find files.",
parameters: {
type: "object",
properties: { paths: { type: "array", items: { type: "string" } } },
required: ["paths"],
},
examples: [{ caption: "Find files", call: { paths: ["src/**/*.ts"] } }],
};
expect(renderToolExamples(tool, "anthropic")).not.toContain("_i");
});
});
+43
View File
@@ -0,0 +1,43 @@
import { describe, expect, it } from "bun:test";
import { z } from "zod/v4";
import { renderToolInventory } from "../src/grammar/inventory";
import type { InbandTool } from "../src/grammar/types";
const searchTool: InbandTool = {
name: "web_search",
description: "Searches the web.",
parameters: z.object({
query: z.string().describe("search query"),
recency: z.enum(["day", "week"]).optional(),
}),
examples: [{ caption: "Basic", call: { query: "rust" } }],
};
describe("renderToolInventory", () => {
it("renders a tool block with a TypeScript signature and native-syntax examples", () => {
const out = renderToolInventory([searchTool], "claude-3-5-sonnet-20241022");
expect(out).toContain("# Tool: web_search");
expect(out).toContain("Searches the web.");
expect(out).toContain("Parameters: {");
expect(out).toContain("query: string;");
expect(out).toContain('recency?: "day" | "week";');
expect(out).toContain("<examples>");
// Examples render in the model's native (anthropic) tool-call syntax.
expect(out).toContain('<invoke name="web_search">');
});
it("omits the examples block when a tool has none", () => {
const tool: InbandTool = {
name: "noop",
description: "No examples.",
parameters: z.object({ x: z.string() }),
};
const out = renderToolInventory([tool], "claude-3-5-sonnet-20241022");
expect(out).toContain("Parameters: {");
expect(out).not.toContain("<examples>");
});
it("returns an empty string when there are no tools", () => {
expect(renderToolInventory([], "claude-3-5-sonnet-20241022")).toBe("");
});
});
+41
View File
@@ -2,6 +2,47 @@
## [Unreleased]
## [15.13.3] - 2026-06-15
### Added
- Added Azure OpenAI as a catalog provider (`azure`, default model `gpt-5.5`, env var `AZURE_OPENAI_API_KEY`), bundling the OpenAI-family models Azure serves over the Responses API (GPT-4/4.1/4o, GPT-5 family, o-series, Codex). Like Amazon Bedrock it is catalog-only — models ship in the bundle and become selectable once the env key is set, with the deployment base URL resolved at runtime from `AZURE_OPENAI_BASE_URL`/`AZURE_OPENAI_RESOURCE_NAME`.
- Added models.dev-backed bundled catalogs for providers that previously shipped no offline models: Hugging Face, Kilo, Moonshot, NanoGPT, Synthetic, Venice, Ollama Cloud, and the Xiaomi Token Plan regions (ams/cn/sgp). They still discover live when credentialed; the bundle is now a non-empty baseline.
### Changed
- Updated stale provider default models to their latest bundled versions: OpenAI-family providers (`azure`, `github-copilot`, `aimlapi`) → GPT-5.5; Gemini providers (`google`, `google-gemini-cli`, `google-vertex`) → `gemini-3.1-pro-preview`; GLM providers (`zai`, `zhipu-coding-plan`) → `glm-5.2`, `cerebras` → `zai-glm-4.7`; Kimi providers (`fireworks`, `opencode-go`, `moonshot`) → `kimi-k2.7-code`, `kimi-code` → `kimi-for-coding`, `together` → `moonshotai/Kimi-K2.7-Code`; `alibaba-coding-plan` → `qwen3.7-plus`; and Claude-Sonnet defaults (`cloudflare-ai-gateway`, `cursor`, `gitlab-duo`, `kilo`, `opencode-zen`, `vercel-ai-gateway`) → Claude Opus 4.x.
- Restricted models.dev Azure discovery to OpenAI-family IDs (`gpt-`, `o1`, `o3`, `o4`, `codex`, `chatgpt`), excluding Foundry-hosted third parties (Claude/DeepSeek/Llama/Mistral/Phi) that Azure serves through non-Responses APIs.
- Detected the Azure OpenAI Responses compat surface (developer role, strict tool mode, strict tool-result pairing) by provider id as well as base URL, so bundled `azure` models whose deployment host is only known at runtime still get the right wire behavior.
- Renamed the `Qwen3-ASR-Flash` model label to `Qwen3 ASR Flash`
### Fixed
- Fixed tool syntax selection for Gemini-family and Gemma model IDs by routing them to dedicated `gemini` and `gemma` formats instead of generic XML
- Fixed `zhipu-coding-plan` and `together` shipping no bundled models: their descriptors referenced non-existent models.dev keys (`zhipu-coding-plan`, `together`); pointed them at the real keys (`zhipuai-coding-plan`, `togetherai`) so they bundle their GLM and full catalogs respectively.
- Folded the `azure-openai-responses` API into the OpenAI Responses thinking-inference branches so Azure reasoning models (o-series, GPT-5, Codex) resolve the discrete effort vocabulary (including `xhigh`) and effort-control mode instead of falling through to generic defaults.
- Fixed `ollama-cloud` discovery inheriting an unsafe cross-provider `contextWindow`/`maxTokens` when `/api/show` returns no size metadata; it now falls back to the safe 128K context / 8K output caps.
- Dropped internal Fireworks control-plane resource ids (`accounts/fireworks/{models,routers}/…`) from the bundle; only the public request ids ship.
## [15.13.2] - 2026-06-15
### Added
- Added the `ToolCallSyntax` union and `FALLBACK_TOOL_SYNTAX` constant to `@oh-my-pi/pi-catalog/identity` (re-exported from `@oh-my-pi/pi-ai/grammar`).
- Added `preferredToolSyntax(modelId)` to `@oh-my-pi/pi-catalog/identity`, resolving a model's native tool-call syntax affinity from its family token (Claude→`anthropic`, GLM→`glm`, Kimi→`kimi`, Qwen→`qwen3`, DeepSeek→`deepseek`, OpenAI/gpt-oss→`harmony`, else the `xml` fallback).
- Added `flux-1-schnell-fp8` to the Fireworks serverless model catalog
- Added `gpt-oss-20b` to the Fireworks model catalog
- Added `qwen3-embedding-8b` to the Fireworks model catalog
- Added `qwen3-reranker-8b` to the Fireworks model catalog
- Added `Gemma 4 E2B IT` and `Gemma 4 E4B IT` to the Google model catalog
- Added `qwen/qwen3-asr-flash` to the Zenmux model catalog
- Added sparse `supportsTools` model metadata so providers can mark models that require in-band tool-call formatting.
### Changed
- Kept non-tool-capable Fireworks serverless models in discovery results and marked them with `supportsTools: false` for fallback-aware handling
- Extended `modelFamilyToken(modelId)` to classify Claude/OpenAI ids the structured parser misses (older dated forms such as `claude-3-5-sonnet-20241022` and `gpt-4o`), returning `anthropic`/`openai` instead of an empty token.
## [15.13.1] - 2026-06-15
### Added
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "15.13.1",
"version": "15.13.3",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -290,6 +290,23 @@ function dropUnusableZaiContextTierIds(models: readonly ModelSpec[]): ModelSpec[
return models.filter(model => !(model.provider === "zai" && model.id.endsWith("[1m]")));
}
/**
* Fireworks discovery and prior snapshots can surface internal control-plane
* resource ids (`accounts/fireworks/{models,routers}/...`) alongside the public
* request ids (`kimi-k2.7-code`, `deepseek-v4-flash`, ...). The wire ids are an
* implementation detail the request path reconstructs from the public id, so
* drop them from the bundle outright.
*/
function dropFireworksWireIds(models: readonly ModelSpec[]): ModelSpec[] {
return models.filter(
model =>
!(
(model.provider === "fireworks" || model.provider === "firepass") &&
model.id.startsWith("accounts/fireworks/")
),
);
}
const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com";
async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise<OAuthAccess | null> {
@@ -477,6 +494,7 @@ async function generateModels() {
allModels = applyCodexPricingFallback(allModels);
allModels = applyFireworksKimiMaxTokensCap(allModels);
allModels = applyFireworksDeepSeekReasoningShape(allModels);
allModels = dropFireworksWireIds(allModels);
allModels = dropUnusableZaiContextTierIds(allModels);
// Normalize display names: gateway author prefixes ("OpenAI: …"), alias
// markers ("(latest)"), provider attribution ("(Antigravity)"), and
+11 -12
View File
@@ -301,29 +301,28 @@ interface OpenAIResponsesSpecLike {
* Build the resolved Responses-API compat record. The Responses flavor
* deliberately differs from chat-completions: GitHub Copilot's responses
* endpoint accepts the `developer` role, while strict tool mode is scoped to
* first-party OpenAI/Azure/Copilot providers. Developer-role and prompt-cache
* detection are URL-only on purpose — the historical call sites never
* consulted the provider id for them. The GPT-5 juice-zero hack keys on the
* model name, matching the historical request-time check.
* first-party OpenAI/Azure/Copilot providers. Azure is detected by provider id
* as well as URL — bundled `azure` models carry no baseUrl (the deployment host
* is per-resource, resolved at runtime) — while OpenAI/Copilot developer-role
* and prompt-cache detection stay URL-keyed, as the historical call sites were.
* The GPT-5 juice-zero hack keys on the model name, matching the historical
* request-time check.
*/
export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): ResolvedOpenAIResponsesCompat {
const baseUrl = spec.baseUrl ?? "";
const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI");
const compat: ResolvedOpenAIResponsesCompat = {
supportsDeveloperRole:
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI") ||
hostMatchesUrl(baseUrl, "githubCopilot"),
supportsDeveloperRole: isAzure || hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "githubCopilot"),
supportsStrictMode:
spec.provider === "openai" ||
spec.provider === "azure" ||
isAzure ||
spec.provider === "github-copilot" ||
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI"),
hostMatchesUrl(baseUrl, "openai"),
supportsReasoningEffort: true,
supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"),
// Azure OpenAI and GitHub Copilot Responses paths require tool results
// to strictly match prior tool calls when building Responses inputs.
strictResponsesPairing: hostMatchesUrl(baseUrl, "azureOpenAI") || spec.provider === "github-copilot",
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"),
reasoningEffortMap: {},
};

Some files were not shown because too many files have changed in this diff Show More