From 2de292632060df509d95fe82c6a7a48ae7b7ad32 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 15 Jun 2026 11:40:55 +0200 Subject: [PATCH] feat: added Gemini and Gemma in-band tool syntax support in runtime - Added Gemini and Gemma syntax routing by model family and owned syntax env values. - Added Gemini and Gemma in-band parsers for tool_code and token-based tool_call streams. - Added rendering support for Gemini fenced tool_code/tool_outputs and Gemma tool tokens. - Fixed parsing edge cases for comments, string escapes, nested args, and truncated blocks. --- docs/toolconv/gemini.md | 145 ++++++ docs/toolconv/gemma.md | 104 +++++ packages/agent/CHANGELOG.md | 5 +- packages/agent/src/agent-loop.ts | 2 + packages/ai/CHANGELOG.md | 10 + packages/ai/src/grammar/factory.ts | 4 + packages/ai/src/grammar/gemini.md | 35 ++ packages/ai/src/grammar/gemini.ts | 440 ++++++++++++++++++ packages/ai/src/grammar/gemma.md | 23 + packages/ai/src/grammar/gemma.ts | 237 ++++++++++ packages/ai/src/grammar/owned-stream.ts | 2 + packages/ai/src/grammar/rendering.ts | 99 ++++ packages/ai/test/gemini-gemma-grammar.test.ts | 205 ++++++++ packages/ai/test/inband-tools.test.ts | 6 + packages/catalog/CHANGELOG.md | 2 + packages/catalog/src/identity/family.ts | 6 + packages/catalog/src/identity/tool-syntax.ts | 17 +- .../test/preferred-tool-syntax.test.ts | 5 +- .../src/modes/components/agent-hub.ts | 1 - 19 files changed, 1344 insertions(+), 4 deletions(-) create mode 100644 docs/toolconv/gemini.md create mode 100644 docs/toolconv/gemma.md create mode 100644 packages/ai/src/grammar/gemini.md create mode 100644 packages/ai/src/grammar/gemini.ts create mode 100644 packages/ai/src/grammar/gemma.md create mode 100644 packages/ai/src/grammar/gemma.ts create mode 100644 packages/ai/test/gemini-gemma-grammar.test.ts diff --git a/docs/toolconv/gemini.md b/docs/toolconv/gemini.md new file mode 100644 index 000000000..f09d6d2cd --- /dev/null +++ b/docs/toolconv/gemini.md @@ -0,0 +1,145 @@ +# Gemini Pythonic tool-calling format (`tool_code` / `default_api`) + +Tool-calling convention of Google's hosted **Gemini** models (current generation, incl. `gemini-3.5-flash` / `*-pro` / `*-preview`) and the **Gemma 3** open-weights family. Both drive tool use **entirely through prompt engineering** — there are **no dedicated special tokens**. The model emits each invocation as **Python source**: a call `default_api.()`, conventionally wrapped in `print(...)` and placed inside a fenced ```` ```tool_code ```` block; it reads results back from a ```` ```tool_outputs ```` block. Because the mechanism is plain text the model was post-trained to produce, the exact same syntax periodically leaks into ordinary output (surfaced by Vertex/AI-Studio as `finish_reason = MALFORMED_FUNCTION_CALL`) — that leak is the clearest public evidence of the format. + +Verified against: the official Gemma 3 function-calling guide (`ai.google.dev/gemma/docs/capabilities/function-calling` — the two recommended prompts, one Pythonic and one JSON), Simon Willison's transcription of those two prompts, Philipp Schmid's Gemma 3 walkthrough (`philschmid.de/gemma-function-calling`), and the reverse-engineered hosted-Gemini form recovered from `MALFORMED_FUNCTION_CALL` reports: `google/adk-go#492` (`Malformed function call: print(default_api.`), `google-gemini/cookbook#929` (`executableCode` part = `print(default_api.get_complaint_number_tool(consumer_number_or_mobile_number='2001234567'))`), `firebase/genkit#2628` (the ```` ```tool_code ```` markdown wrapper), and the Google AI dev-forum thread "Gemini 2 flash returns raw markdown instead of function call" (71964). + +## "Special" tokens + +**None.** Nothing here is a control token in the tokenizer's special-token table — every marker below BPE-splits into ordinary text and survives a `skip_special_tokens=True` decode. This is the defining property of the convention and the reason it both (a) works across hosted Gemini and open Gemma without tokenizer support and (b) leaks. The functional markers are: + +| Marker (verbatim) | Role | +|---|---| +| ` ```tool_code ` | Opens a fenced block whose body is Python the app must execute. Closed by a bare ` ``` `. | +| ` ```tool_outputs ` | Opens a fenced block carrying the executed results back to the model. Closed by a bare ` ``` `. | +| `default_api` | Synthetic module namespace the hosted stack bundles un-namespaced tools into. Calls read `default_api.(...)`. | +| `print(...)` | Conventional wrapper around the call in the hosted-Gemini form (the model is trained to "print" the call). Semantically irrelevant — the runtime parses the call, it does not execute Python. | + +There is **no** per-call id on the wire and **no** in-band reasoning marker — Gemini reasoning travels out of band as API "thought signatures", never as ``-style text. + +## Roles / turn structure + +The Pythonic payload is independent of the envelope, and the envelope differs by deployment: + +- **Hosted Gemini** uses the normal `contents[]` turn structure (`role: "user" | "model"`); the `tool_code` block appears inside a `model` turn's text, and `tool_outputs` is supplied as the next turn. +- **Gemma 3** (open weights) uses the Gemma chat template (`user … ` / `model`); the tool prompt is prepended to the first user turn and the blocks live inside model/user turns. + +This document specifies the **payload** (the two fenced blocks + the Python call form); the surrounding turn tokens belong to whichever template hosts it. + +## Tool definitions + +Tools are advertised in the prompt as a JSON-Schema catalog. Gemma 3's official guide ships **two** interchangeable system-prompt templates that differ only in how the model is told to answer: + +1. **Pythonic** (the one this spec targets): + > You have access to functions. If you decide to invoke any of the function(s), you MUST put it in the format of `[func_name1(params_name1=params_value1, params_name2=params_value2...), func_name2(params)]` + > You SHOULD NOT include any other text in the response if you call a function + +2. **JSON** (the sibling convention — see `qwen3.md` for the closely related Hermes shape): + > … you MUST put it in the format of `{"name": function name, "parameters": dictionary of argument name and its value}` + +Hosted Gemini wraps the same idea in markdown fences and the `default_api` namespace. The function signatures themselves are passed as OpenAI-style tool JSON (`{"type":"function","function":{name,description,parameters}}`). + +## Tool-call format + +One call is a Python call expression. The hosted-Gemini canonical form is a `print()` of a `default_api` method, fenced: + +````text +```tool_code +print(default_api.get_current_temperature(location="London", unit="celsius")) +``` +```` + +All of the following are accepted equivalents seen in the wild and across Gemma/Gemini variants; a robust parser normalizes them to `{name, arguments}`: + +- `print(default_api.NAME(KWARGS))` — hosted Gemini canonical. +- `default_api.NAME(KWARGS)` — `print`/namespace are optional sugar. +- `NAME(KWARGS)` — bare call (Gemma 3 Pythonic prompt). +- `result = NAME(KWARGS)` — assignment form (Gemma 3 docs use `result = convert(...)`). + +Argument values are **Python literals**, not JSON: + +| Python literal | Example | Decoded | +|---|---|---| +| string | `'London'` or `"London"` | `"London"` | +| int / float | `42`, `3.14` | `42`, `3.14` | +| bool | `True` / `False` | `true` / `false` | +| null | `None` | `null` | +| list | `["a", "b"]` | `["a","b"]` | +| dict | `{"k": 1}` | `{"k":1}` | + +Strings use Python escaping (`\n`, `\t`, `\\`, `\'`, `\"`); hosted Gemini emits single quotes (`location='London'`), Gemma examples use double quotes — both are valid. Arguments are keyword form (`name=value`); positional arguments are not used because the runtime maps to a named schema. + +## Multiple / parallel tool calls + +Two encodings exist, both inside a single `tool_code` block: + +- **Gemma 3 Pythonic prompt** — a Python **list** of call expressions: + ````text + ```tool_code + [get_current_temperature(location="London"), get_temperature_date(location="London", date="2024-10-01")] + ``` + ```` +- **Hosted Gemini** — one `print(default_api...)` **statement per line**: + ````text + ```tool_code + print(default_api.get_current_temperature(location="London")) + print(default_api.get_temperature_date(location="London", date="2024-10-01")) + ``` + ```` + +Either way the calls are returned in source order; the application executes them and returns one result per call in the same order. + +## Tool-result format + +Executed results are returned to the model in a ```` ```tool_outputs ```` block. Gemma 3 docs use assignment-style values (`result = 92.3`); for opaque tool output the block simply carries the returned text/JSON: + +````text +```tool_outputs +{"temperature": 26.1, "location": "London", "unit": "celsius"} +``` +```` + +The model then continues with either a natural-language answer or another `tool_code` block. + +## End-to-end example + +````text + +What's the temperature in London? + + +```tool_code +print(default_api.get_current_temperature(location="London", unit="celsius")) +``` + + +```tool_outputs +{"temperature": 11.4, "location": "London", "unit": "celsius"} +``` + + +It's currently 11.4°C in London. +```` + +## OpenAI-compatible / native API mapping + +- Hosted Gemini's native API normally returns a structured `functionCall` part (`{name, args}`); for Gemini 3 each carries an `id` that must be echoed in the matching `functionResponse`, plus a `thoughtSignature` that must be preserved. The Pythonic text form is what you get when the structured path *fails* (`finish_reason = MALFORMED_FUNCTION_CALL`) or when tool use is driven purely by prompt (Gemma, or Gemini via the code-execution `executableCode` part). +- When parsed out of an OpenAI-compatible shim, each recovered call becomes `tool_calls[i] = {id (server-minted), type:"function", function:{name, arguments:}}` — the Python kwargs are re-serialized to a JSON string at that boundary. +- Feed results back as the deployment's tool/`functionResponse` turn (hosted) or a `tool_outputs` block in the next user turn (prompt-driven). + +## Parsing notes & gotchas + +- **Python, not JSON.** `True`/`False`/`None` (not `true`/`false`/`null`), single-quoted strings, and trailing commas are all legal. A JSON parser will reject valid calls; decode Python literals. +- **Strip the wrapper.** Normalize away `print(...)`, a `default_api.` (or any `module.`) prefix, and an `LHS =` assignment before reading the call name. `print` is never a tool name. +- **Skip string contents when scanning.** A call like `search(pattern="foo(")` contains a `(` inside a string; a naive `\w+\(` scan mis-detects `foo` as a callee. Track string state and only treat top-level `(` as a call opener. +- **Fence ambiguity.** The body terminates at the first bare ` ``` `; a string argument literally containing ` ``` ` will truncate the block early (rare, accepted limitation). +- **It leaks.** Because nothing is a special token, the format appears verbatim in normal responses when the model "decides" to call a tool but the structured decoder misfires. Production code reading raw text should detect ` ```tool_code ` and parse it; production code on the structured API should retry on `MALFORMED_FUNCTION_CALL`. +- **Variant divergence.** Gemma **4** abandoned this Pythonic form for a token-delimited brace syntax (`<|tool_call>call:NAME{…}`) — a different convention documented in `gemma.md`. This spec covers hosted Gemini and Gemma 3. + +## Sources + +- Gemma 3 function calling (two recommended prompts): https://ai.google.dev/gemma/docs/capabilities/function-calling +- Simon Willison, "Function calling with Gemma": https://simonwillison.net/2025/Mar/26/function-calling-with-gemma/ +- Philipp Schmid, "Google Gemma 3 Function Calling Example": https://www.philschmid.de/gemma-function-calling +- Gemini 3 thought signatures + functionCall ids: https://ai.google.dev/gemini-api/docs/gemini-3 +- `default_api` / `tool_code` leak evidence: https://github.com/google/adk-go/issues/492 · https://github.com/google-gemini/cookbook/issues/929 · https://github.com/firebase/genkit/issues/2628 · https://discuss.ai.google.dev/t/gemini-2-flash-api-returns-raw-markdown-instead-of-function-call/71964 diff --git a/docs/toolconv/gemma.md b/docs/toolconv/gemma.md new file mode 100644 index 000000000..6cf7cd3ad --- /dev/null +++ b/docs/toolconv/gemma.md @@ -0,0 +1,104 @@ +# Gemma 4 tool-calling format (token-delimited `call:NAME{…}`) + +Tool-calling convention of Google's **Gemma 4** open-weights family (`google/gemma-4-*-it`). It is a clean break from the prompt-engineered Pythonic `tool_code` form used by Gemma 3 and hosted Gemini (see `gemini.md`): Gemma 4 introduces **dedicated special tokens** and a compact **token-delimited brace syntax**. Tool declarations, calls, and responses each get their own paired markers, and every string value is wrapped in a `<|"|>` token rather than ASCII quotes. The model emits one call as `<|tool_call>call:NAME{key:value,…}`; the developer parses it, runs the tool, and appends `<|tool_response>response:NAME{…}`. + +Verified against: the official "Function calling with Gemma 4" guide (`ai.google.dev/gemma/docs/capabilities/text/function-calling-gemma4`), including the byte-exact `processor.apply_chat_template(...)` renderings and the reference `extract_tool_calls` regex it ships. All example streams below are copied from that page (model `google/gemma-4-E2B-it`). + +## Special tokens + +Gemma 4 wraps each structural element in a paired token. Note the **asymmetric pipe placement** — an opener carries the pipe on the left (`<|x>`) and its closer carries it on the right (``): + +| Open | Close | Purpose | +|---|---|---| +| `` | — | Beginning of sequence | +| `<|turn>` | `` | One conversation turn; the role name is the first line of the body | +| `<|tool>` | `` | A tool **declaration** block (in the system turn) | +| `<|tool_call>` | `` | One tool **call** emitted by the model | +| `<|tool_response>` | `` | One tool **result** fed back to the model | +| `<|"|>` | `<|"|>` | String-literal delimiter (same token on both ends) | +| `` | — | End of sequence | + +Because the string delimiter is a token (`<|"|>`), values may contain raw ASCII quotes and commas without escaping — only a literal `<|"|>` token sequence cannot appear inside a string. + +## Roles / turn structure + +Each turn is `<|turn>{role}\n{body}`. Roles are `system`, `user`, `model`. With a generation prompt the stream ends at `<|turn>model\n` and the model continues. Tool declarations are merged into the `system` turn; tool calls and the following tool responses are emitted inside the `model` turn (the response block immediately follows the call block in the re-rendered history). + +## Tool definitions + +Each tool is declared in the system turn as `<|tool>declaration:NAME{…}`, where the body is the schema serialized in the same brace syntax used by calls. Types are upper-cased strings (`STRING`, `OBJECT`, …). Byte-exact, from the guide: + +```text +<|tool>declaration:get_current_temperature{description:<|"|>Gets the current temperature for a given location.<|"|>,parameters:{properties:{location:{description:<|"|>The city name, e.g. San Francisco<|"|>,type:<|"|>STRING<|"|>} },required:[<|"|>location<|"|>],type:<|"|>OBJECT<|"|>} } +``` + +## Tool-call format + +The model emits one call per `<|tool_call>…` block. The body is `call:NAME{ARGS}`, where `ARGS` is a comma-separated list of `key:value` pairs: + +```text +<|tool_call>call:get_current_temperature{location:<|"|>London<|"|>} +``` + +Value grammar inside `{…}`: + +| Value kind | Encoding | Example | +|---|---|---| +| string | `<|"|>text<|"|>` | `location:<|"|>London<|"|>` | +| int / float | bare | `count:42` | +| bool | bare | `flag:true` | +| null | bare | `unit:null` | +| list | `[v,v,…]` | `tags:[<|"|>a<|"|>,<|"|>b<|"|>]` | +| nested object | `{k:v,…}` | `config:{theme:<|"|>dark<|"|>}` | + +The reference parser shipped in the guide: + +```python +[{ + "name": name, + "arguments": { + k: cast((v1 or v2).strip()) + for k, v1, v2 in re.findall(r'(\w+):(?:<\|"\|>(.*?)<\|"\|>|([^,}]*))', args) + } +} for name, args in re.findall(r"<\|tool_call>call:(\w+)\{(.*?)\}", text, re.DOTALL)] +``` + +i.e. each argument value is either a `<|"|>…<|"|>` string or a bare run of non-`,}` characters (cast to int/float/bool, else kept as a string). + +## Multiple / parallel tool calls + +Parallel calls are consecutive `<|tool_call>…` blocks (one call each), returned in order. The application returns one `<|tool_response>` per call in the same order. + +## Tool-result format + +Each result is `<|tool_response>response:NAME{…}`, the response object serialized in the same brace syntax. Byte-exact, from the guide's re-rendered history: + +```text +<|tool_response>response:get_current_weather{temperature:15,weather:<|"|>sunny<|"|>} +``` + +## End-to-end example + +Byte-exact `apply_chat_template` output from the guide (system + tool, user, model call, tool response, final answer — note the response block sits in the same model turn, right after the call): + +```text +<|turn>system +You are a helpful assistant.<|tool>declaration:get_current_weather{description:<|"|>Gets the current weather in a given location.<|"|>,parameters:{properties:{location:{description:<|"|>The city and state, e.g. "San Francisco, CA" or "Tokyo, JP"<|"|>,type:<|"|>STRING<|"|>},unit:{description:<|"|>The unit to return the temperature in.<|"|>,enum:[<|"|>celsius<|"|>,<|"|>fahrenheit<|"|>],type:<|"|>STRING<|"|>} },required:[<|"|>location<|"|>],type:<|"|>OBJECT<|"|>} } +<|turn>user +Hey, what's the weather in Tokyo right now? +<|turn>model +<|tool_call>call:get_current_weather{location:<|"|>Tokyo, JP<|"|>}<|tool_response>response:get_current_weather{temperature:15,weather:<|"|>sunny<|"|>}The current weather in Tokyo is 15 degrees Celsius and sunny. +``` + +## Parsing notes & gotchas + +- **String delimiter is a token, not a quote.** Inside `<|"|>…<|"|>` the bytes `"` and `,` are literal data — the example `<|"|>The city and state, e.g. "San Francisco, CA"…<|"|>` contains both. Split arguments on `,`/`}` only **outside** a `<|"|>…<|"|>` span. +- **Asymmetric pipes.** The closer is ``, not `` or `<|tool_call>`. Matching the wrong pipe side will never close the block. +- **One call per block.** Unlike a JSON `tool_calls[]` array, parallelism is "more blocks", not "more entries in one block". +- **Bare scalars.** A value not wrapped in `<|"|>` is `true`/`false` → bool, `null`/`none` → null, numeric → number, otherwise a bare string (e.g. an unquoted enum or type name like `STRING`). +- **Not Gemma 3 / hosted Gemini.** Those use the Pythonic `tool_code` / `default_api` form in `gemini.md`. Gemma 4 replaced it with this token syntax; the two are not interchangeable. + +## Sources + +- Function calling with Gemma 4 (byte-exact chat-template renderings + reference parser): https://ai.google.dev/gemma/docs/capabilities/text/function-calling-gemma4 +- Gemma 4 prompt formatting: https://ai.google.dev/gemma/docs/core/prompt-formatting-gemma4 diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 16b62066d..596a0f08c 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +### Added + +- Added support for `gemini` and `gemma` as valid owned tool syntax values in environment configuration ## [15.13.2] - 2026-06-15 @@ -803,4 +806,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon ### Changed - `Agent` constructor now has all options optional (empty options use defaults). -- `queueMessage()` is now synchronous (no longer returns a Promise). +- `queueMessage()` is now synchronous (no longer returns a Promise). \ No newline at end of file diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index ac4644baa..5f85d357e 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -98,6 +98,8 @@ function resolveOwnedToolSyntaxFromEnv(value: string | undefined): ToolCallSynta case "harmony": case "pi": case "qwen3": + case "gemini": + case "gemma": return value; default: return undefined; diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index abc815cc5..69f2b6d03 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,10 +1,20 @@ # Changelog ## [Unreleased] + ### Added +- Added the `gemini` in-band tool-call syntax with Python-style ```tool_code``` blocks and `default_api` invocations +- Added the `gemma` token-delimited in-band tool-call syntax using `<|tool_call>` and `<|tool_response>` blocks +- Added `gemini` and `gemma` to owned stream tool-result token detection so their tool responses are recognized +- Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming - Added the Azure OpenAI provider definition (`azure`) to the registry; `AZURE_OPENAI_API_KEY` resolves as its env-var API key via the catalog provider table. +### Fixed + +- Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming +- Fixed Gemini/Gemma in-band tool-call parsing around Python comments, raw/unicode string literals, and Gemma close-token text inside string values. + ## [15.13.2] - 2026-06-15 ### Added diff --git a/packages/ai/src/grammar/factory.ts b/packages/ai/src/grammar/factory.ts index 7f31a12c0..af6efb96f 100644 --- a/packages/ai/src/grammar/factory.ts +++ b/packages/ai/src/grammar/factory.ts @@ -1,5 +1,7 @@ import anthropicGrammar from "./anthropic"; import deepseekGrammar from "./deepseek"; +import geminiGrammar from "./gemini"; +import gemmaGrammar from "./gemma"; import glmGrammar from "./glm"; import harmonyGrammar from "./harmony"; import hermesGrammar from "./hermes"; @@ -19,6 +21,8 @@ const GRAMMARS: Record = { harmony: harmonyGrammar, pi: piGrammar, qwen3: qwen3Grammar, + gemini: geminiGrammar, + gemma: gemmaGrammar, }; export function getInbandGrammar(syntax: ToolCallSyntax): Grammar { diff --git a/packages/ai/src/grammar/gemini.md b/packages/ai/src/grammar/gemini.md new file mode 100644 index 000000000..d83bd693d --- /dev/null +++ b/packages/ai/src/grammar/gemini.md @@ -0,0 +1,35 @@ +## Format guide + +Emit tool calls as Python inside a fenced ` ```tool_code ` block. Call each function as a method on `default_api`: + +````text +```tool_code +default_api.function_name(arg="value", count=2) +``` +```` + +Argument values are Python literals: `"strings"`, numbers, `True`/`False`, `None`, `[lists]`, `{"dicts": 1}`. + +Call several functions in parallel as a Python list: + +````text +```tool_code +[default_api.first(x="a"), default_api.second(y="b")] +``` +```` + +Tool results arrive later in a ` ```tool_outputs ` block: + +````text +```tool_outputs +verbatim tool result +``` +```` + +## Rules + +- The function name MUST match a listed function; arguments are keyword form (`name=value`). +- Multiple calls = a single `[...]` list (or one `default_api...` call per line) inside one ` ```tool_code ` block. +- Put any reasoning as plain text before the ` ```tool_code ` block, never inside it. +- Read each ` ```tool_outputs ` block in call order. NEVER write a ` ```tool_outputs ` block yourself. +- After emitting the ` ```tool_code ` block, YOU MUST STOP AND HALT. diff --git a/packages/ai/src/grammar/gemini.ts b/packages/ai/src/grammar/gemini.ts new file mode 100644 index 000000000..5131a9615 --- /dev/null +++ b/packages/ai/src/grammar/gemini.ts @@ -0,0 +1,440 @@ +import { mintToolCallId, partialSuffixOverlapAny } from "./coercion"; +import grammarPrompt from "./gemini.md" with { type: "text" }; +import { renderGeminiInvocation, renderGeminiToolCalls, renderGeminiToolResults } from "./rendering"; +import type { Grammar, InbandScanEvent, InbandScanner } from "./types"; + +const CODE_OPEN = "```tool_code"; +const FENCE = "```"; +const OPEN_TAGS = [CODE_OPEN] as const; + +type State = "outside" | "tool"; + +interface ParsedCall { + name: string; + arguments: Record; +} + +/** + * Scanner for the hosted-Gemini / Gemma 3 Pythonic tool-calling convention + * (see `docs/toolconv/gemini.md`). Tool calls arrive as a ```` ```tool_code ```` + * fenced block whose body is one or more Python call expressions, e.g. + * `print(default_api.search(pattern="x", skip=40))`. Like the qwen3 scanner we + * buffer the whole block until its closing fence, then parse all calls at once + * (no incremental argument deltas — Python literals are not worth streaming). + */ +export class GeminiInbandScanner implements InbandScanner { + #buffer = ""; + #state: State = "outside"; + + feed(text: string): InbandScanEvent[] { + if (text.length === 0) return []; + this.#buffer += text; + return this.#consume(false); + } + + flush(): InbandScanEvent[] { + return this.#consume(true); + } + + #consume(final: boolean): InbandScanEvent[] { + const events: InbandScanEvent[] = []; + while (this.#buffer.length > 0) { + if (this.#state === "outside") { + this.#consumeOutside(final, events); + if (this.#state === "outside") break; + continue; + } + this.#consumeTool(final, events); + if (this.#state === "tool") break; + } + return events; + } + + #consumeOutside(final: boolean, events: InbandScanEvent[]): void { + const open = this.#buffer.indexOf(CODE_OPEN); + if (open === -1) { + const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OPEN_TAGS); + const emit = this.#buffer.slice(0, this.#buffer.length - hold); + if (emit.length > 0) events.push({ type: "text", text: emit }); + this.#buffer = this.#buffer.slice(this.#buffer.length - hold); + return; + } + if (open > 0) events.push({ type: "text", text: this.#buffer.slice(0, open) }); + this.#buffer = this.#buffer.slice(open + CODE_OPEN.length); + this.#state = "tool"; + } + + #consumeTool(final: boolean, events: InbandScanEvent[]): void { + const close = this.#buffer.indexOf(FENCE); + if (close === -1) { + // Inside the fence we emit nothing until it closes; on a truncated + // stream the incomplete block is dropped rather than leaked as text. + if (final) { + this.#buffer = ""; + this.#state = "outside"; + } + return; + } + const body = this.#buffer.slice(0, close); + const rawBlock = `${CODE_OPEN}${body}${FENCE}`; + for (const call of parseGeminiCalls(body)) { + const id = mintToolCallId(); + events.push({ type: "toolStart", id, name: call.name }); + events.push({ type: "toolEnd", id, name: call.name, arguments: call.arguments, rawBlock }); + } + this.#buffer = this.#buffer.slice(close + FENCE.length); + this.#state = "outside"; + } +} + +/** Extract every top-level call expression in a `tool_code` body. */ +function parseGeminiCalls(body: string): ParsedCall[] { + const calls: ParsedCall[] = []; + let i = 0; + const n = body.length; + while (i < n) { + const ch = body[i]!; + if (ch === '"' || ch === "'") { + i = skipString(body, i); + continue; + } + if (ch === "#") { + i = skipComment(body, i); + continue; + } + if (ch === "(") { + const name = identBefore(body, i); + if (name && name !== "print") { + const end = matchParen(body, i); + if (end !== -1) { + calls.push({ name, arguments: parsePyArgs(body.slice(i + 1, end)) }); + i = end + 1; + continue; + } + } + } + i++; + } + return calls; +} + +/** Identifier immediately preceding a `(` (the callee's final name segment). */ +function identBefore(body: string, parenIndex: number): string | undefined { + let j = parenIndex - 1; + while (j >= 0 && /\s/.test(body[j]!)) j--; + const end = j + 1; + while (j >= 0 && /[A-Za-z0-9_]/.test(body[j]!)) j--; + const name = body.slice(j + 1, end); + return /^[A-Za-z_]\w*$/.test(name) ? name : undefined; +} + +/** Index of the `)` matching the `(` at `openIndex`, skipping string contents. */ +function matchParen(body: string, openIndex: number): number { + let depth = 0; + let i = openIndex; + const n = body.length; + while (i < n) { + const ch = body[i]!; + if (ch === '"' || ch === "'") { + i = skipString(body, i); + continue; + } + if (ch === "#") { + i = skipComment(body, i); + continue; + } + if (ch === "(") depth++; + else if (ch === ")" && --depth === 0) return i; + i++; + } + return -1; +} + +/** Index just past the Python string literal starting at `i` (a quote char). */ +function skipString(body: string, i: number): number { + const quote = body[i]!; + const triple = quote + quote + quote; + if (body.startsWith(triple, i)) { + const close = body.indexOf(triple, i + 3); + return close === -1 ? body.length : close + 3; + } + let j = i + 1; + const n = body.length; + while (j < n) { + const ch = body[j]!; + if (ch === "\\") { + j += 2; + continue; + } + if (ch === quote) return j + 1; + j++; + } + return n; +} + +function skipComment(body: string, i: number): number { + const newline = body.indexOf("\n", i + 1); + return newline === -1 ? body.length : newline + 1; +} + +function stripComments(body: string): string { + let out = ""; + let i = 0; + const n = body.length; + while (i < n) { + const ch = body[i]!; + if (ch === '"' || ch === "'") { + const end = skipString(body, i); + out += body.slice(i, end); + i = end; + continue; + } + if (ch === "#") { + const newline = body.indexOf("\n", i + 1); + if (newline === -1) break; + out += "\n"; + i = newline + 1; + continue; + } + out += ch; + i++; + } + return out; +} + +function parsePyArgs(text: string): Record { + const out: Record = {}; + for (const segment of splitTopLevel(stripComments(text), ",")) { + const trimmed = segment.trim(); + if (trimmed.length === 0) continue; + const eq = topLevelIndexOf(trimmed, "="); + if (eq === -1) continue; // positional args are not part of the convention + const key = trimmed.slice(0, eq).trim(); + if (!/^[A-Za-z_]\w*$/.test(key)) continue; + out[key] = parsePyValue(trimmed.slice(eq + 1).trim()); + } + return out; +} + +function parsePyValue(raw: string): unknown { + const t = raw.trim(); + if (t.length === 0) return ""; + if (t === "True" || t === "true") return true; + if (t === "False" || t === "false") return false; + if (t === "None" || t === "null") return null; + const prefix = stringPrefixLength(t); + if (prefix !== undefined) return decodeString(t); + const first = t[0]!; + if (first === "[") return parseList(t); + if (first === "{") return parseDict(t); + if (/^[+-]?(\d|\.)/.test(t)) { + const num = Number(t); + if (!Number.isNaN(num)) return num; + } + return t; +} + +function parseList(t: string): unknown[] { + const inner = t.slice(1, t.endsWith("]") ? t.length - 1 : t.length); + return splitTopLevel(stripComments(inner), ",") + .map(part => part.trim()) + .filter(part => part.length > 0) + .map(parsePyValue); +} + +function parseDict(t: string): Record { + const inner = t.slice(1, t.endsWith("}") ? t.length - 1 : t.length); + const out: Record = {}; + for (const segment of splitTopLevel(stripComments(inner), ",")) { + const trimmed = segment.trim(); + if (trimmed.length === 0) continue; + const colon = topLevelIndexOf(trimmed, ":"); + if (colon === -1) continue; + const keyRaw = trimmed.slice(0, colon).trim(); + const key = stringPrefixLength(keyRaw) !== undefined ? decodeString(keyRaw) : keyRaw; + out[key] = parsePyValue(trimmed.slice(colon + 1).trim()); + } + return out; +} + +function decodeString(t: string): string { + const prefix = stringPrefixLength(t) ?? 0; + const raw = t.slice(0, prefix).toLowerCase().includes("r"); + const quote = t[prefix]!; + const triple = quote + quote + quote; + if (t.startsWith(triple, prefix) && t.length >= prefix + 6 && t.endsWith(triple)) { + const inner = t.slice(prefix + 3, t.length - 3); + return raw ? inner : unescapePythonString(inner); + } + const inner = t.endsWith(quote) && t.length >= prefix + 2 ? t.slice(prefix + 1, t.length - 1) : t.slice(prefix + 1); + return raw ? inner : unescapePythonString(inner); +} + +function stringPrefixLength(t: string): number | undefined { + for (const len of [2, 1, 0]) { + const prefix = t.slice(0, len).toLowerCase(); + if ( + (prefix === "" || prefix === "r" || prefix === "u" || prefix === "b" || prefix === "br" || prefix === "rb") && + (t[len] === '"' || t[len] === "'") + ) { + return len; + } + } + return undefined; +} + +function unescapePythonString(s: string): string { + if (!s.includes("\\")) return s; + let out = ""; + let i = 0; + while (i < s.length) { + const ch = s[i]!; + if (ch !== "\\") { + out += ch; + i++; + continue; + } + const next = s[i + 1]; + if (next && /^[0-7]$/.test(next)) { + const octal = /^[0-7]{1,3}/.exec(s.slice(i + 1))![0]; + out += String.fromCharCode(parseInt(octal, 8)); + i += octal.length + 1; + continue; + } + switch (next) { + case "n": + out += "\n"; + i += 2; + break; + case "t": + out += "\t"; + i += 2; + break; + case "r": + out += "\r"; + i += 2; + break; + case "\\": + out += "\\"; + i += 2; + break; + case "'": + out += "'"; + i += 2; + break; + case '"': + out += '"'; + i += 2; + break; + case "0": + out += "\0"; + i += 2; + break; + case "x": { + const hex = s.slice(i + 2, i + 4); + if (/^[0-9a-fA-F]{2}$/.test(hex)) { + out += String.fromCharCode(parseInt(hex, 16)); + i += 4; + } else { + out += "x"; + i += 2; + } + break; + } + case "u": { + const hex = s.slice(i + 2, i + 6); + if (/^[0-9a-fA-F]{4}$/.test(hex)) { + out += String.fromCharCode(parseInt(hex, 16)); + i += 6; + } else { + out += "u"; + i += 2; + } + break; + } + case "U": { + const hex = s.slice(i + 2, i + 10); + if (/^[0-9a-fA-F]{8}$/.test(hex)) { + out += String.fromCodePoint(parseInt(hex, 16)); + i += 10; + } else { + out += "U"; + i += 2; + } + break; + } + case undefined: + out += "\\"; + i += 1; + break; + default: + out += next; + i += 2; + break; + } + } + return out; +} + +/** Split on `sep` at bracket depth 0, skipping string literals. */ +function splitTopLevel(text: string, sep: string): string[] { + const parts: string[] = []; + let depth = 0; + let start = 0; + let i = 0; + const n = text.length; + while (i < n) { + const ch = text[i]!; + if (ch === '"' || ch === "'") { + i = skipString(text, i); + continue; + } + if (ch === "#") { + i = skipComment(text, i); + continue; + } + if (ch === "(" || ch === "[" || ch === "{") depth++; + else if (ch === ")" || ch === "]" || ch === "}") depth--; + else if (depth === 0 && ch === sep) { + parts.push(text.slice(start, i)); + start = i + 1; + } + i++; + } + parts.push(text.slice(start)); + return parts; +} + +/** First index of `ch` at bracket depth 0, skipping string literals. */ +function topLevelIndexOf(text: string, ch: string): number { + let depth = 0; + let i = 0; + const n = text.length; + while (i < n) { + const c = text[i]!; + if (c === '"' || c === "'") { + i = skipString(text, i); + continue; + } + if (c === "#") { + i = skipComment(text, i); + continue; + } + if (c === "(" || c === "[" || c === "{") depth++; + else if (c === ")" || c === "]" || c === "}") depth--; + else if (depth === 0 && c === ch) return i; + i++; + } + return -1; +} + +const grammar: Grammar = { + syntax: "gemini", + prompt: grammarPrompt, + createScanner: () => new GeminiInbandScanner(), + renderToolCall: renderGeminiInvocation, + renderAssistantToolCalls: renderGeminiToolCalls, + renderToolResults: renderGeminiToolResults, +}; + +export default grammar; diff --git a/packages/ai/src/grammar/gemma.md b/packages/ai/src/grammar/gemma.md new file mode 100644 index 000000000..821dd9a8b --- /dev/null +++ b/packages/ai/src/grammar/gemma.md @@ -0,0 +1,23 @@ +## Format guide + +Emit each tool call as one `<|tool_call>` block. The body is `call:NAME{key:value,...}`; wrap every string value in the `<|"|>` token: + +```text +<|tool_call>call:function_name{path:<|"|>src/a.ts<|"|>,count:2} +``` + +Non-string values are bare: numbers (`2`), `true`/`false`, `null`, lists `[<|"|>a<|"|>,<|"|>b<|"|>]`, and nested objects `{k:<|"|>v<|"|>}`. + +Tool results arrive later in matching `<|tool_response>` blocks: + +```text +<|tool_response>response:function_name{output:<|"|>verbatim result<|"|>} +``` + +## Rules + +- `NAME` MUST match a listed function; arguments are `key:value` pairs separated by commas. +- Multiple calls = consecutive `<|tool_call>...` blocks; keep prose outside them. +- The closer is `` (pipe on the right), not `` or `<|tool_call>`. +- Read each `<|tool_response>` block in call order. NEVER write a `<|tool_response>` block yourself. +- After emitting your tool calls, YOU MUST STOP AND HALT. diff --git a/packages/ai/src/grammar/gemma.ts b/packages/ai/src/grammar/gemma.ts new file mode 100644 index 000000000..fd34600ec --- /dev/null +++ b/packages/ai/src/grammar/gemma.ts @@ -0,0 +1,237 @@ +import { mintToolCallId, partialSuffixOverlapAny } from "./coercion"; +import grammarPrompt from "./gemma.md" with { type: "text" }; +import { renderGemmaInvocation, renderGemmaToolCalls, renderGemmaToolResults } from "./rendering"; +import type { Grammar, InbandScanEvent, InbandScanner } from "./types"; + +const CALL_OPEN = "<|tool_call>"; +const CALL_CLOSE = ""; +const STRING = '<|"|>'; +const OPEN_TAGS = [CALL_OPEN] as const; +const CALL_HEAD = /^call:\s*([A-Za-z_]\w*)\s*\{/; + +type State = "outside" | "tool"; + +interface ParsedCall { + name: string; + arguments: Record; +} + +/** + * Scanner for the Gemma 4 token-delimited tool-calling convention (see + * `docs/toolconv/gemma.md`). Each call is one `<|tool_call>call:NAME{…}` + * block whose argument list is `key:value` pairs; string values are wrapped in + * the `<|"|>` token rather than ASCII quotes, so splitting must skip those spans. + */ +export class GemmaInbandScanner implements InbandScanner { + #buffer = ""; + #state: State = "outside"; + + feed(text: string): InbandScanEvent[] { + if (text.length === 0) return []; + this.#buffer += text; + return this.#consume(false); + } + + flush(): InbandScanEvent[] { + return this.#consume(true); + } + + #consume(final: boolean): InbandScanEvent[] { + const events: InbandScanEvent[] = []; + while (this.#buffer.length > 0) { + if (this.#state === "outside") { + this.#consumeOutside(final, events); + if (this.#state === "outside") break; + continue; + } + this.#consumeTool(final, events); + if (this.#state === "tool") break; + } + return events; + } + + #consumeOutside(final: boolean, events: InbandScanEvent[]): void { + const open = this.#buffer.indexOf(CALL_OPEN); + if (open === -1) { + const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OPEN_TAGS); + const emit = this.#buffer.slice(0, this.#buffer.length - hold); + if (emit.length > 0) events.push({ type: "text", text: emit }); + this.#buffer = this.#buffer.slice(this.#buffer.length - hold); + return; + } + if (open > 0) events.push({ type: "text", text: this.#buffer.slice(0, open) }); + this.#buffer = this.#buffer.slice(open + CALL_OPEN.length); + this.#state = "tool"; + } + + #consumeTool(final: boolean, events: InbandScanEvent[]): void { + const close = findCallClose(this.#buffer); + if (close === -1) { + if (final) { + this.#buffer = ""; + this.#state = "outside"; + } + return; + } + const body = this.#buffer.slice(0, close); + const parsed = parseGemmaCall(body); + if (parsed) { + const id = mintToolCallId(); + events.push({ type: "toolStart", id, name: parsed.name }); + events.push({ + type: "toolEnd", + id, + name: parsed.name, + arguments: parsed.arguments, + rawBlock: `${CALL_OPEN}${body}${CALL_CLOSE}`, + }); + } + this.#buffer = this.#buffer.slice(close + CALL_CLOSE.length); + this.#state = "outside"; + } +} + +function parseGemmaCall(body: string): ParsedCall | undefined { + const trimmed = body.trim(); + const head = CALL_HEAD.exec(trimmed); + if (!head) return undefined; + const braceStart = head[0].length - 1; + const end = matchDelim(trimmed, braceStart, "{", "}"); + const argsText = end === -1 ? trimmed.slice(braceStart + 1) : trimmed.slice(braceStart + 1, end); + return { name: head[1]!, arguments: parseGemmaArgs(argsText) }; +} + +function parseGemmaArgs(text: string): Record { + const out: Record = {}; + for (const segment of splitTopLevel(text, ",")) { + const trimmed = segment.trim(); + if (trimmed.length === 0) continue; + const colon = topLevelIndexOf(trimmed, ":"); + if (colon === -1) continue; + const key = trimmed.slice(0, colon).trim(); + if (!/^[A-Za-z_]\w*$/.test(key)) continue; + out[key] = parseGemmaValue(trimmed.slice(colon + 1).trim()); + } + return out; +} + +function parseGemmaValue(raw: string): unknown { + const t = raw.trim(); + if (t.startsWith(STRING)) { + const close = t.indexOf(STRING, STRING.length); + return close === -1 ? t.slice(STRING.length) : t.slice(STRING.length, close); + } + if (t.startsWith("[")) { + const end = matchDelim(t, 0, "[", "]"); + const inner = end === -1 ? t.slice(1) : t.slice(1, end); + return splitTopLevel(inner, ",") + .map(part => part.trim()) + .filter(part => part.length > 0) + .map(parseGemmaValue); + } + if (t.startsWith("{")) { + const end = matchDelim(t, 0, "{", "}"); + return parseGemmaArgs(end === -1 ? t.slice(1) : t.slice(1, end)); + } + if (t === "true") return true; + if (t === "false") return false; + if (t === "null" || t === "none" || t === "None") return null; + if (/^[+-]?(\d|\.)/.test(t)) { + const num = Number(t); + if (!Number.isNaN(num)) return num; + } + return t; +} + +/** Index just past the `<|"|>`-delimited string starting at `i`. */ +function skipGemmaString(text: string, i: number): number { + const close = text.indexOf(STRING, i + STRING.length); + return close === -1 ? text.length : close + STRING.length; +} + +function findCallClose(text: string): number { + let i = 0; + const n = text.length; + while (i < n) { + if (text.startsWith(STRING, i)) { + i = skipGemmaString(text, i); + continue; + } + if (text.startsWith(CALL_CLOSE, i)) return i; + i++; + } + return -1; +} + +/** Index of the `close` delimiter matching `open` at `openIndex`, skipping strings. */ +function matchDelim(text: string, openIndex: number, open: string, close: string): number { + let depth = 0; + let i = openIndex; + const n = text.length; + while (i < n) { + if (text.startsWith(STRING, i)) { + i = skipGemmaString(text, i); + continue; + } + const ch = text[i]!; + if (ch === open) depth++; + else if (ch === close && --depth === 0) return i; + i++; + } + return -1; +} + +/** Split on `sep` at bracket depth 0, skipping `<|"|>` string spans. */ +function splitTopLevel(text: string, sep: string): string[] { + const parts: string[] = []; + let depth = 0; + let start = 0; + let i = 0; + const n = text.length; + while (i < n) { + if (text.startsWith(STRING, i)) { + i = skipGemmaString(text, i); + continue; + } + const ch = text[i]!; + if (ch === "{" || ch === "[" || ch === "(") depth++; + else if (ch === "}" || ch === "]" || ch === ")") depth--; + else if (depth === 0 && ch === sep) { + parts.push(text.slice(start, i)); + start = i + 1; + } + i++; + } + parts.push(text.slice(start)); + return parts; +} + +/** First index of `ch` at bracket depth 0, skipping `<|"|>` string spans. */ +function topLevelIndexOf(text: string, ch: string): number { + let depth = 0; + let i = 0; + const n = text.length; + while (i < n) { + if (text.startsWith(STRING, i)) { + i = skipGemmaString(text, i); + continue; + } + const c = text[i]!; + if (c === "{" || c === "[" || c === "(") depth++; + else if (c === "}" || c === "]" || c === ")") depth--; + else if (depth === 0 && c === ch) return i; + i++; + } + return -1; +} + +const grammar: Grammar = { + syntax: "gemma", + prompt: grammarPrompt, + createScanner: () => new GemmaInbandScanner(), + renderToolCall: renderGemmaInvocation, + renderAssistantToolCalls: renderGemmaToolCalls, + renderToolResults: renderGemmaToolResults, +}; + +export default grammar; diff --git a/packages/ai/src/grammar/owned-stream.ts b/packages/ai/src/grammar/owned-stream.ts index 906d59d28..634afc2da 100644 --- a/packages/ai/src/grammar/owned-stream.ts +++ b/packages/ai/src/grammar/owned-stream.ts @@ -20,6 +20,8 @@ const RESPONSE_OPEN_TOKENS: Record = { harmony: ["<|start|>functions."], pi: [""], qwen3: [""], + gemini: ["```tool_outputs"], + gemma: ["<|tool_response>"], }; function firstTokenIndex(text: string, tokens: readonly string[]): number { diff --git a/packages/ai/src/grammar/rendering.ts b/packages/ai/src/grammar/rendering.ts index 64a5ac041..b8967f720 100644 --- a/packages/ai/src/grammar/rendering.ts +++ b/packages/ai/src/grammar/rendering.ts @@ -212,3 +212,102 @@ function escapeXmlAttr(value: string): string { function escapeXmlText(value: string): string { return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">"); } + +// --- Gemini: Pythonic `tool_code` / `default_api` convention --- + +const GEMINI_CODE_OPEN = "```tool_code"; +const GEMINI_OUTPUT_OPEN = "```tool_outputs"; +const GEMINI_FENCE = "```"; + +export function renderGeminiInvocation(call: ToolCall, _options: GrammarRenderOptions = {}): string { + const kwargs = Object.entries(call.arguments) + .map(([key, value]) => `${key}=${pyValue(value)}`) + .join(", "); + return `default_api.${call.name}(${kwargs})`; +} + +export function renderGeminiToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string { + // One call renders bare; parallel calls render as a Python list `[a, b]`. + const body = + calls.length === 1 + ? renderGeminiInvocation(calls[0]!, options) + : `[${calls.map(call => renderGeminiInvocation(call, options)).join(", ")}]`; + // Examples show the bare call; the live wire form fences it as `tool_code`. + return options.example ? body : `${GEMINI_CODE_OPEN}\n${body}\n${GEMINI_FENCE}`; +} + +export function renderGeminiToolResults(results: readonly GrammarToolResult[]): string { + return results.map(result => `${GEMINI_OUTPUT_OPEN}\n${result.text}\n${GEMINI_FENCE}`).join("\n"); +} + +function pyValue(value: unknown): string { + if (value === null || value === undefined) return "None"; + if (typeof value === "boolean") return value ? "True" : "False"; + if (typeof value === "number") return Number.isFinite(value) ? String(value) : pyString(String(value)); + if (typeof value === "string") return pyString(value); + if (Array.isArray(value)) return `[${value.map(pyValue).join(", ")}]`; + if (typeof value === "object") { + const entries = Object.entries(value as Record); + return `{${entries.map(([key, val]) => `${pyString(key)}: ${pyValue(val)}`).join(", ")}}`; + } + return pyString(String(value)); +} + +function pyString(value: string): string { + const escaped = value + .replaceAll("\\", "\\\\") + .replaceAll('"', '\\"') + .replaceAll("\n", "\\n") + .replaceAll("\r", "\\r") + .replaceAll("\t", "\\t"); + return `"${escaped}"`; +} + +// --- Gemma 4: token-delimited `call:NAME{…}` convention --- + +const GEMMA_CALL_OPEN = "<|tool_call>"; +const GEMMA_CALL_CLOSE = ""; +const GEMMA_RESPONSE_OPEN = "<|tool_response>"; +const GEMMA_RESPONSE_CLOSE = ""; +const GEMMA_STRING = '<|"|>'; + +export function renderGemmaInvocation(call: ToolCall, _options: GrammarRenderOptions = {}): string { + const args = Object.entries(call.arguments) + .map(([key, value]) => `${key}:${gemmaValue(value)}`) + .join(","); + return `${GEMMA_CALL_OPEN}call:${call.name}{${args}}${GEMMA_CALL_CLOSE}`; +} + +export function renderGemmaToolCalls(calls: readonly ToolCall[], options: GrammarRenderOptions = {}): string { + return calls.map(call => renderGemmaInvocation(call, options)).join(""); +} + +export function renderGemmaToolResults(results: readonly GrammarToolResult[]): string { + return results + .map( + result => + `${GEMMA_RESPONSE_OPEN}response:${result.name}{output:${gemmaValue(parseMaybeJson(result.text))}}${GEMMA_RESPONSE_CLOSE}`, + ) + .join(""); +} + +function gemmaValue(value: unknown): string { + if (value === null || value === undefined) return "null"; + if (typeof value === "boolean") return value ? "true" : "false"; + if (typeof value === "number") return String(value); + if (typeof value === "string") return `${GEMMA_STRING}${value}${GEMMA_STRING}`; + if (Array.isArray(value)) return `[${value.map(gemmaValue).join(",")}]`; + if (typeof value === "object") { + const entries = Object.entries(value as Record); + return `{${entries.map(([key, val]) => `${key}:${gemmaValue(val)}`).join(",")}}`; + } + return `${GEMMA_STRING}${String(value)}${GEMMA_STRING}`; +} + +function parseMaybeJson(text: string): unknown { + try { + return JSON.parse(text) as unknown; + } catch { + return text; + } +} diff --git a/packages/ai/test/gemini-gemma-grammar.test.ts b/packages/ai/test/gemini-gemma-grammar.test.ts new file mode 100644 index 000000000..34d9bf73b --- /dev/null +++ b/packages/ai/test/gemini-gemma-grammar.test.ts @@ -0,0 +1,205 @@ +import { describe, expect, it } from "bun:test"; +import type { ToolCall } from "@oh-my-pi/pi-ai"; +import { + createInbandScanner, + getInbandGrammar, + type InbandScanEvent, + type ToolCallSyntax, +} from "@oh-my-pi/pi-ai/grammar"; + +function scan(syntax: ToolCallSyntax, text: string, charByChar = false): InbandScanEvent[] { + const scanner = createInbandScanner(syntax); + const events: InbandScanEvent[] = []; + if (charByChar) for (const ch of text) events.push(...scanner.feed(ch)); + else events.push(...scanner.feed(text)); + events.push(...scanner.flush()); + return events; +} + +function parsedCalls( + syntax: ToolCallSyntax, + text: string, + charByChar = false, +): { name: string; arguments: Record }[] { + return scan(syntax, text, charByChar) + .filter((event): event is Extract => event.type === "toolEnd") + .map(event => ({ name: event.name, arguments: event.arguments })); +} + +function visibleText(events: readonly InbandScanEvent[]): string { + return events + .filter((event): event is Extract => event.type === "text") + .map(event => event.text) + .join(""); +} + +const call = (name: string, args: Record): ToolCall => ({ + type: "toolCall", + id: name, + name, + arguments: args, +}); + +describe("gemini grammar (Pythonic tool_code)", () => { + it("parses the print(default_api...) form", () => { + const calls = parsedCalls("gemini", "```tool_code\nprint(default_api.read(path='a.ts', count=2))\n```"); + expect(calls).toEqual([{ name: "read", arguments: { path: "a.ts", count: 2 } }]); + }); + + it("parses bare default_api calls and the assignment form", () => { + expect(parsedCalls("gemini", '```tool_code\ndefault_api.search(pattern="x")\n```')).toEqual([ + { name: "search", arguments: { pattern: "x" } }, + ]); + expect(parsedCalls("gemini", '```tool_code\nresult = search(pattern="x")\n```')).toEqual([ + { name: "search", arguments: { pattern: "x" } }, + ]); + }); + + it("decodes Python literals (bool/None/number/list/dict)", () => { + const calls = parsedCalls( + "gemini", + '```tool_code\ndefault_api.f(s="hi", n=3, r=1.5, b=True, z=None, arr=[1, 2], obj={"k": "v"})\n```', + ); + expect(calls[0]!.arguments).toEqual({ s: "hi", n: 3, r: 1.5, b: true, z: null, arr: [1, 2], obj: { k: "v" } }); + }); + + it("ignores parens and commas inside string arguments", () => { + const calls = parsedCalls("gemini", '```tool_code\ndefault_api.search(pattern="foo(a, b)", flag=False)\n```'); + expect(calls[0]!.arguments).toEqual({ pattern: "foo(a, b)", flag: false }); + }); + + it("ignores Python comments and decodes raw/unicode string literals", () => { + const text = [ + "```tool_code", + '# default_api.write(path="ignored")', + "result = read(", + ' path=r"src/(foo)\\.ts", # default_api.write(path="ignored")', + " count=2,", + ' meta={"emoji": "\\U0001F600"},', + ")", + '[default_api.write(path="out", content="foo(,bar")]', + "```", + ].join("\n"); + + const calls = parsedCalls("gemini", text); + + expect(calls.map(parsed => parsed.name)).toEqual(["read", "write"]); + expect(calls[0]!.arguments).toEqual({ path: "src/(foo)\\.ts", count: 2, meta: { emoji: "😀" } }); + expect(calls[1]!.arguments).toEqual({ path: "out", content: "foo(,bar" }); + }); + + it("parses parallel calls written as a [a, b] list", () => { + const calls = parsedCalls( + "gemini", + '```tool_code\n[default_api.read(path="a"), default_api.write(path="b", content="c")]\n```', + ); + expect(calls).toEqual([ + { name: "read", arguments: { path: "a" } }, + { name: "write", arguments: { path: "b", content: "c" } }, + ]); + }); + + it("preserves prose outside the fence", () => { + const text = visibleText(scan("gemini", 'before\n```tool_code\ndefault_api.read(path="a")\n```\nafter')); + expect(text).toContain("before"); + expect(text).toContain("after"); + expect(text).not.toContain("default_api"); + }); + + it("yields the same calls when streamed character by character", () => { + const text = '```tool_code\ndefault_api.read(path="a.ts", count=7)\n```'; + expect(parsedCalls("gemini", text, true)).toEqual([{ name: "read", arguments: { path: "a.ts", count: 7 } }]); + }); + + it("renders parallel calls as a list and round-trips through the scanner", () => { + const grammar = getInbandGrammar("gemini"); + const rendered = grammar.renderAssistantToolCalls([ + call("read", { path: "a" }), + call("write", { path: "b", content: "c" }), + ]); + expect(rendered).toContain("```tool_code"); + expect(rendered).toContain("[default_api.read(path="); + expect(rendered).toContain("default_api.write(path="); + expect(parsedCalls("gemini", rendered)).toEqual([ + { name: "read", arguments: { path: "a" } }, + { name: "write", arguments: { path: "b", content: "c" } }, + ]); + }); + + it("renders examples without a fence or print wrapper", () => { + const grammar = getInbandGrammar("gemini"); + expect(grammar.renderToolCall(call("read", { path: "a.ts" }), { example: true })).toBe( + 'default_api.read(path="a.ts")', + ); + expect(grammar.renderAssistantToolCalls([call("read", { path: "a.ts" })], { example: true })).toBe( + 'default_api.read(path="a.ts")', + ); + }); + + it("escapes special characters on render and decodes them on parse", () => { + const grammar = getInbandGrammar("gemini"); + const rendered = grammar.renderAssistantToolCalls([call("write", { content: 'a "b"\n\tc\\d' })]); + expect(parsedCalls("gemini", rendered)).toEqual([{ name: "write", arguments: { content: 'a "b"\n\tc\\d' } }]); + }); +}); + +describe("gemma grammar (token-delimited call:NAME{…})", () => { + it("parses a single call with string and scalar args", () => { + const calls = parsedCalls("gemma", '<|tool_call>call:read{path:<|"|>a.ts<|"|>,count:2}'); + expect(calls).toEqual([{ name: "read", arguments: { path: "a.ts", count: 2 } }]); + }); + + it('keeps commas and quotes inside <|"|> string values', () => { + const calls = parsedCalls("gemma", '<|tool_call>call:f{loc:<|"|>San Francisco, CA "downtown"<|"|>}'); + expect(calls[0]!.arguments).toEqual({ loc: 'San Francisco, CA "downtown"' }); + }); + + it("keeps close-token text inside string values", () => { + const calls = parsedCalls( + "gemma", + '<|tool_call>call:read{path:<|"|>literal marker, ok<|"|>,count:2}', + true, + ); + + expect(calls).toEqual([{ name: "read", arguments: { path: "literal marker, ok", count: 2 } }]); + }); + + it("parses scalars, lists, and nested objects", () => { + const calls = parsedCalls( + "gemma", + '<|tool_call>call:f{b:true,z:null,n:3,arr:[<|"|>a<|"|>,<|"|>b<|"|>],obj:{k:<|"|>v<|"|>}}', + ); + expect(calls[0]!.arguments).toEqual({ b: true, z: null, n: 3, arr: ["a", "b"], obj: { k: "v" } }); + }); + + it("parses consecutive blocks as parallel calls", () => { + const calls = parsedCalls( + "gemma", + '<|tool_call>call:read{path:<|"|>a<|"|>}<|tool_call>call:write{path:<|"|>b<|"|>}', + ); + expect(calls).toEqual([ + { name: "read", arguments: { path: "a" } }, + { name: "write", arguments: { path: "b" } }, + ]); + }); + + it("yields the same call when streamed character by character", () => { + const text = '<|tool_call>call:read{path:<|"|>a.ts<|"|>,count:7}'; + expect(parsedCalls("gemma", text, true)).toEqual([{ name: "read", arguments: { path: "a.ts", count: 7 } }]); + }); + + it("renders calls that round-trip through the scanner", () => { + const grammar = getInbandGrammar("gemma"); + const rendered = grammar.renderAssistantToolCalls([ + call("read", { path: "a" }), + call("write", { path: "b", content: "c" }), + ]); + expect(rendered).toBe( + '<|tool_call>call:read{path:<|"|>a<|"|>}<|tool_call>call:write{path:<|"|>b<|"|>,content:<|"|>c<|"|>}', + ); + expect(parsedCalls("gemma", rendered)).toEqual([ + { name: "read", arguments: { path: "a" } }, + { name: "write", arguments: { path: "b", content: "c" } }, + ]); + }); +}); diff --git a/packages/ai/test/inband-tools.test.ts b/packages/ai/test/inband-tools.test.ts index c5c024eae..588792482 100644 --- a/packages/ai/test/inband-tools.test.ts +++ b/packages/ai/test/inband-tools.test.ts @@ -42,6 +42,8 @@ const SYNTAXES: readonly ToolCallSyntax[] = [ "harmony", "pi", "qwen3", + "gemini", + "gemma", ]; function usage(): Usage { @@ -205,6 +207,10 @@ describe("in-band tool grammars", () => { "\nFILE\n", ); expect(getInbandGrammar("pi").renderToolResults([resultBlock])).toBe("\nFILE\n"); + expect(getInbandGrammar("gemini").renderToolResults([resultBlock])).toBe("```tool_outputs\nFILE\n```"); + expect(getInbandGrammar("gemma").renderToolResults([resultBlock])).toBe( + '<|tool_response>response:read{output:<|"|>FILE<|"|>}', + ); }); it("encodes assistant calls and tool results through the selected grammar", () => { diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index c8901db14..09d54e806 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Added - Added Azure OpenAI as a catalog provider (`azure`, default model `gpt-5.5`, env var `AZURE_OPENAI_API_KEY`), bundling the OpenAI-family models Azure serves over the Responses API (GPT-4/4.1/4o, GPT-5 family, o-series, Codex). Like Amazon Bedrock it is catalog-only — models ship in the bundle and become selectable once the env key is set, with the deployment base URL resolved at runtime from `AZURE_OPENAI_BASE_URL`/`AZURE_OPENAI_RESOURCE_NAME`. @@ -15,6 +16,7 @@ ### Fixed +- Fixed tool syntax selection for Gemini-family and Gemma model IDs by routing them to dedicated `gemini` and `gemma` formats instead of generic XML - Fixed `zhipu-coding-plan` and `together` shipping no bundled models: their descriptors referenced non-existent models.dev keys (`zhipu-coding-plan`, `together`); pointed them at the real keys (`zhipuai-coding-plan`, `togetherai`) so they bundle their GLM and full catalogs respectively. - Folded the `azure-openai-responses` API into the OpenAI Responses thinking-inference branches so Azure reasoning models (o-series, GPT-5, Codex) resolve the discrete effort vocabulary (including `xhigh`) and effort-control mode instead of falling through to generic defaults. - Fixed `ollama-cloud` discovery inheriting an unsafe cross-provider `contextWindow`/`maxTokens` when `/api/show` returns no size metadata; it now falls back to the safe 128K context / 8K output caps. diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index 1381344eb..eb3fbf06b 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -41,6 +41,11 @@ export function isQwenModelId(modelId: string): boolean { return modelId.toLowerCase().includes("qwen"); } +/** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */ +export function isGemmaModelId(modelId: string): boolean { + return /(^|\/)gemma[-.]?\d/i.test(modelId); +} + /** DeepSeek family by id or display name (proxies often rename the id but keep the name). */ export function isDeepseekModelIdOrName(value: string): boolean { return value.toLowerCase().includes("deepseek"); @@ -127,6 +132,7 @@ export function modelFamilyToken(modelId: string): string { if (isOpenAIGptOssModelId(modelId)) return "gpt-oss"; if (isDeepseekModelIdOrName(modelId)) return "deepseek"; if (isMimoModelIdOrName(modelId)) return "mimo"; + if (isGemmaModelId(modelId)) return "gemma"; if (parseGlmModel(bareModelId(modelId))) return "glm"; return ""; } diff --git a/packages/catalog/src/identity/tool-syntax.ts b/packages/catalog/src/identity/tool-syntax.ts index 0149ec0d2..74933ef67 100644 --- a/packages/catalog/src/identity/tool-syntax.ts +++ b/packages/catalog/src/identity/tool-syntax.ts @@ -1,6 +1,17 @@ import { modelFamilyToken } from "./family"; -export type ToolCallSyntax = "glm" | "hermes" | "kimi" | "xml" | "anthropic" | "deepseek" | "harmony" | "pi" | "qwen3"; +export type ToolCallSyntax = + | "glm" + | "hermes" + | "kimi" + | "xml" + | "anthropic" + | "deepseek" + | "harmony" + | "pi" + | "qwen3" + | "gemini" + | "gemma"; export const FALLBACK_TOOL_SYNTAX: ToolCallSyntax = "xml"; @@ -10,6 +21,10 @@ export function preferredToolSyntax(modelId: string): ToolCallSyntax { return "anthropic"; case "glm": return "glm"; + case "gemini": + return "gemini"; + case "gemma": + return "gemma"; case "kimi": return "kimi"; case "qwen": diff --git a/packages/catalog/test/preferred-tool-syntax.test.ts b/packages/catalog/test/preferred-tool-syntax.test.ts index 5b372e16c..2085a2d65 100644 --- a/packages/catalog/test/preferred-tool-syntax.test.ts +++ b/packages/catalog/test/preferred-tool-syntax.test.ts @@ -10,7 +10,10 @@ describe("preferredToolSyntax", () => { expect(preferredToolSyntax("qwen-coder-32b-instruct")).toBe("qwen3"); expect(preferredToolSyntax("gpt-4o-mini")).toBe("harmony"); expect(preferredToolSyntax("gpt-oss-120b")).toBe("harmony"); - expect(preferredToolSyntax("gemini-1.5-pro")).toBe("xml"); + expect(preferredToolSyntax("gemini-1.5-pro")).toBe("gemini"); + expect(preferredToolSyntax("gemini-3.5-flash")).toBe("gemini"); + expect(preferredToolSyntax("gemma-3-27b-it")).toBe("gemma"); + expect(preferredToolSyntax("google/gemma-4-E2B-it")).toBe("gemma"); expect(preferredToolSyntax("unclassified-model-id")).toBe("xml"); }); }); diff --git a/packages/coding-agent/src/modes/components/agent-hub.ts b/packages/coding-agent/src/modes/components/agent-hub.ts index ec85d6889..c193e6359 100644 --- a/packages/coding-agent/src/modes/components/agent-hub.ts +++ b/packages/coding-agent/src/modes/components/agent-hub.ts @@ -196,7 +196,6 @@ export class AgentHubOverlayComponent extends Container { /** Captured row order from the first refresh; keeps the hub stable while open. */ #rowOrder: Map | undefined; - // Chat state #chatAgentId: string | undefined; #editor: Editor;