fix(ai): routed llama.cpp parallel tool calls by item.call_id
llama.cpp's /v1/responses emits function_call `output_item.added` with
only `item.call_id` (no `item.id`, no `output_index`); the matching
`function_call_arguments.delta` carries `item_id: "fc_<call_id>"`.
`processResponsesStream` keyed its lookup map by `item.id` only, so
the registry stayed empty and every delta fell back to `lastOpenItem`
— the most recently added block. With N parallel calls, N-1 of them
finalized with arguments `{}` and the agent rejected them with
`path: Invalid input: expected string, received undefined`, while the
trailing call hoarded every delta.
Register function-call and custom-tool-call items under `item.call_id`
as a secondary key alongside `item.id`/`output_index`, and look up by
the same fallback in `output_item.done`. Real OpenAI is unaffected
(distinct keys, both populated). Regression test pins the exact
llama.cpp emission shape.
Fixes #2015
This commit is contained in:
@@ -17,6 +17,7 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)).
|
||||
- Fixed parallel `function_call` items losing arguments against llama.cpp's OpenAI Responses endpoint (`/v1/responses`), where every call but the last finalized with `{}` and the agent rejected them with `path: Invalid input: expected string, received undefined`. llama.cpp's `to_json_oaicompat_resp` emits `output_item.added` with only `item.call_id` (no `item.id`, no `output_index`) while the matching `function_call_arguments.delta` carries `item_id: "fc_<call_id>"`. `processResponsesStream` now registers function-call and custom-tool-call items under `item.call_id` as a secondary lookup key (alongside `item.id`/`output_index`) so identifier-deviant hosts route deltas and done events to the right block. ([#2015](https://github.com/can1357/oh-my-pi/issues/2015))
|
||||
|
||||
## [15.9.67] - 2026-06-06
|
||||
|
||||
|
||||
@@ -459,6 +459,13 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
// see https://github.com/can1357/oh-my-pi/issues/1880 — llama.cpp emits parallel
|
||||
// function_call deltas interleaved, and a singleton `current` reference would
|
||||
// fold them into the wrong block and drop arguments on every call but the last.
|
||||
//
|
||||
// llama.cpp's `to_json_oaicompat_resp` (issue #2015) compounds this: `output_item.added`
|
||||
// for function_call/custom_tool_call carries `item.call_id` but no `item.id` and no
|
||||
// `output_index`, while the matching `function_call_arguments.delta` carries
|
||||
// `item_id = "fc_<call_id>"`. Registering function-call items by `call_id` as a
|
||||
// secondary key lets the delta lookup find the right block on hosts that emit one
|
||||
// identifier but not the other.
|
||||
const openItemsByOutputIndex = new Map<number, StreamingItem>();
|
||||
const openItemsByItemId = new Map<string, StreamingItem>();
|
||||
let lastOpenItem: StreamingItem | null = null;
|
||||
@@ -468,9 +475,11 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
outputIndex: number | undefined,
|
||||
itemId: string | undefined,
|
||||
entry: StreamingItem,
|
||||
alternateItemKey?: string,
|
||||
): void => {
|
||||
if (typeof outputIndex === "number") openItemsByOutputIndex.set(outputIndex, entry);
|
||||
if (itemId) openItemsByItemId.set(itemId, entry);
|
||||
if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.set(alternateItemKey, entry);
|
||||
openItemsInOrder.push(entry);
|
||||
lastOpenItem = entry;
|
||||
};
|
||||
@@ -508,9 +517,11 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
outputIndex: number | undefined,
|
||||
itemId: string | undefined,
|
||||
entry: StreamingItem | undefined,
|
||||
alternateItemKey?: string,
|
||||
): void => {
|
||||
if (typeof outputIndex === "number") openItemsByOutputIndex.delete(outputIndex);
|
||||
if (itemId) openItemsByItemId.delete(itemId);
|
||||
if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.delete(alternateItemKey);
|
||||
if (entry) {
|
||||
const index = openItemsInOrder.indexOf(entry);
|
||||
if (index >= 0) openItemsInOrder.splice(index, 1);
|
||||
@@ -550,7 +561,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
partialJson: item.arguments || "",
|
||||
};
|
||||
output.content.push(block);
|
||||
registerOpenItem(event.output_index, item.id, { item, block });
|
||||
registerOpenItem(event.output_index, item.id, { item, block }, item.call_id);
|
||||
stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output });
|
||||
} else if (item.type === "custom_tool_call") {
|
||||
const block: StreamingToolCallBlock = {
|
||||
@@ -568,7 +579,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
partialJson: item.input ?? "",
|
||||
};
|
||||
output.content.push(block);
|
||||
registerOpenItem(event.output_index, item.id, { item, block });
|
||||
registerOpenItem(event.output_index, item.id, { item, block }, item.call_id);
|
||||
stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output });
|
||||
}
|
||||
} else if (event.type === "response.reasoning_summary_part.added") {
|
||||
@@ -709,7 +720,10 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
} else if (event.type === "response.output_item.done") {
|
||||
const item = structuredCloneJSON(event.item);
|
||||
options?.onOutputItemDone?.(item);
|
||||
const entry = lookupOpenItem({ output_index: event.output_index, item_id: item.id });
|
||||
const entry =
|
||||
item.type === "function_call" || item.type === "custom_tool_call"
|
||||
? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id })
|
||||
: lookupOpenItem({ output_index: event.output_index, item_id: item.id });
|
||||
if (item.type === "reasoning") {
|
||||
const thinking =
|
||||
item.summary?.length > 0
|
||||
@@ -768,7 +782,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
delete (block as { argumentsDone?: boolean }).argumentsDone;
|
||||
}
|
||||
const contentIndex = block ? contentIndexOf(block) : output.content.length - 1;
|
||||
closeOpenItem(event.output_index, item.id, entry);
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id);
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
} else if (item.type === "custom_tool_call") {
|
||||
const block = entry?.block.type === "toolCall" ? entry.block : undefined;
|
||||
@@ -781,7 +795,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
customWireName: item.name,
|
||||
};
|
||||
const contentIndex = block ? contentIndexOf(block) : output.content.length - 1;
|
||||
closeOpenItem(event.output_index, item.id, entry);
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id);
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
}
|
||||
} else if (event.type === "response.completed") {
|
||||
|
||||
@@ -260,4 +260,79 @@ describe("processResponsesStream: parallel function_call items", () => {
|
||||
expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ command: "printf a" });
|
||||
expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ command: "printf b" });
|
||||
});
|
||||
|
||||
test("routes deltas by item.call_id when llama.cpp omits item.id and output_index (issue #2015)", async () => {
|
||||
// llama.cpp's `to_json_oaicompat_resp` (tools/server/server-task.cpp) emits a
|
||||
// function_call's `output_item.added` with only `item.call_id` — no `item.id`,
|
||||
// no `output_index`. The matching `function_call_arguments.delta` then carries
|
||||
// `item_id: "fc_<call_id>"` and again no `output_index`. Without secondary
|
||||
// indexing on `call_id`, `processResponsesStream`'s lookup map stays empty and
|
||||
// every delta lands on the trailing block, leaving earlier calls with empty
|
||||
// arguments (= `{}`) — the read tool then rejects them with
|
||||
// `path: Invalid input: expected string, received undefined`.
|
||||
const output = makeOutput();
|
||||
const emitted: EmittedEvent[] = [];
|
||||
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
|
||||
|
||||
const argsA = JSON.stringify({ path: "a.txt" });
|
||||
const argsB = JSON.stringify({ path: "b.txt" });
|
||||
const argsC = JSON.stringify({ path: "c.txt" });
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", call_id: "fc_a", name: "read", arguments: "" },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", call_id: "fc_b", name: "read", arguments: "" },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", call_id: "fc_c", name: "read", arguments: "" },
|
||||
},
|
||||
{ type: "response.function_call_arguments.delta", item_id: "fc_a", delta: argsA },
|
||||
{ type: "response.function_call_arguments.delta", item_id: "fc_b", delta: argsB },
|
||||
{ type: "response.function_call_arguments.delta", item_id: "fc_c", delta: argsC },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", call_id: "fc_a", name: "read", arguments: argsA },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", call_id: "fc_b", name: "read", arguments: argsB },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", call_id: "fc_c", name: "read", arguments: argsC },
|
||||
},
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.content).toHaveLength(3);
|
||||
const [a, b, c] = output.content;
|
||||
if (a?.type !== "toolCall" || b?.type !== "toolCall" || c?.type !== "toolCall") {
|
||||
throw new Error("expected toolCalls");
|
||||
}
|
||||
expect(a.arguments).toEqual({ path: "a.txt" });
|
||||
expect(b.arguments).toEqual({ path: "b.txt" });
|
||||
expect(c.arguments).toEqual({ path: "c.txt" });
|
||||
|
||||
const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{
|
||||
toolCall: { id: string; arguments: Record<string, unknown> };
|
||||
contentIndex: number;
|
||||
}>;
|
||||
expect(ends).toHaveLength(3);
|
||||
const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e]));
|
||||
expect(byCallId.get("fc_a")?.toolCall.arguments).toEqual({ path: "a.txt" });
|
||||
expect(byCallId.get("fc_b")?.toolCall.arguments).toEqual({ path: "b.txt" });
|
||||
expect(byCallId.get("fc_c")?.toolCall.arguments).toEqual({ path: "c.txt" });
|
||||
expect(byCallId.get("fc_a")?.contentIndex).toBe(0);
|
||||
expect(byCallId.get("fc_b")?.contentIndex).toBe(1);
|
||||
expect(byCallId.get("fc_c")?.contentIndex).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user