diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 728c8ddff..e2079a3ee 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -22,6 +22,9 @@ - Improved connection error handling by classifying generic connection failures as transient, allowing them to be retried, while keeping explicit authentication rejections non-retryable. - Fixed custom Anthropic base URLs losing native thinking signatures during continuation requests. - Fixed Alibaba Coding Plan Custom login rejecting valid API keys on endpoints that do not serve the default validation model by validating against the model catalog instead. +### Fixed + +- Fixed OpenAI Responses token-cap truncations suppressing fully streamed function and custom tool calls whose inputs are complete. ## [17.0.6] - 2026-07-20 diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index b2e564586..32112b8f3 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -2501,6 +2501,7 @@ export async function processResponsesStream( const entry = lookupOpenToolCallAlias(event, "custom_tool_call"); if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") { finalizeCustomToolCallInputDone(entry.block, event.input); + entry.block[kStreamingArgumentsDone] = true; } } else if (event.type === "response.output_item.done") { const item = structuredCloneJSON(event.item); @@ -2620,6 +2621,10 @@ export async function processResponsesStream( } } else if (terminalEvent) { const response = terminalEvent.response; + const shouldPromoteIncompleteToolUse = + response?.status === "incomplete" && + response.incomplete_details?.reason === "max_output_tokens" && + hasExecutableIncompleteResponsesToolCalls(output); finalizePendingResponsesToolCalls(output); if (response?.id) { output.responseId = response.id; @@ -2656,7 +2661,11 @@ export async function processResponsesStream( kind: "content-blocked", }); } - promoteResponsesToolUseStopReason(output, (response as { end_turn?: boolean } | undefined)?.end_turn); + promoteResponsesToolUseStopReason( + output, + (response as { end_turn?: boolean } | undefined)?.end_turn, + shouldPromoteIncompleteToolUse, + ); options?.onCompleted?.(); // `response.completed`/`response.incomplete`/`response.done` is the last event of a // Responses stream. Stop pulling instead of waiting for the server to @@ -2712,6 +2721,28 @@ export function mapOpenAIResponsesStopReason(status: ResponseStatus | undefined) } } +function hasExecutableIncompleteResponsesToolCalls(output: AssistantMessage): boolean { + let hasToolCall = false; + for (const block of output.content) { + if (block.type !== "toolCall") continue; + hasToolCall = true; + const pending = block as ToolCall & { + [kStreamingPartialJson]?: string; + [kStreamingArgumentsDone]?: boolean; + }; + const rawArguments = pending[kStreamingPartialJson]; + // `output_item.done` is not positive completion proof: our Responses + // compatibility encoder force-closes still-open calls before forwarding an + // upstream `length` stop. Only an explicit arguments/input-done event sets + // this marker; an open ordinary call can instead prove completion with its + // retained strict-complete JSON. + if (pending[kStreamingArgumentsDone]) continue; + if (pending.customWireName !== undefined || rawArguments === undefined) return false; + if (classifyJsonPrefix(rawArguments) !== "complete") return false; + } + return hasToolCall; +} + /** * Finalize any streamed toolCall block whose `output_item.done` never arrived * (lossy proxy, or a terminal event that raced the per-item done): parse the @@ -2745,8 +2776,15 @@ export function finalizePendingResponsesToolCalls(output: AssistantMessage): voi * re-samples instead of ending. Callers set `output.stopReason` from the wire * status first via {@link mapOpenAIResponsesStopReason}. */ -export function promoteResponsesToolUseStopReason(output: AssistantMessage, endTurn: boolean | undefined): void { - if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") { +export function promoteResponsesToolUseStopReason( + output: AssistantMessage, + endTurn: boolean | undefined, + promoteIncompleteToolUse = false, +): void { + if ( + output.content.some(block => block.type === "toolCall") && + (output.stopReason === "stop" || (promoteIncompleteToolUse && output.stopReason === "length")) + ) { output.stopReason = "toolUse"; } if (endTurn === false && output.stopReason === "stop") { diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index d7f80ffe5..8b9f2e4b4 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -1,9 +1,9 @@ // Terminal-event contracts for `processResponsesStream`: // // 1. `response.incomplete` is a terminal frame (max_output_tokens / content -// filter truncation). It must populate usage and map to stopReason -// "length" — previously it was ignored entirely, so truncated responses -// reported stopReason "stop" with zero usage and no cost. +// filter truncation). It must populate usage and normally map to stopReason +// "length", while a fully streamed function call remains executable as +// "toolUse". // 2. `response.output_item.done` for a custom_tool_call must persist the final // input on the stored content block and drop the transient `partialJson` // accumulation buffer, mirroring the function_call branch. @@ -120,6 +120,325 @@ describe("processResponsesStream: terminal events", () => { expect(output.content).toEqual([expect.objectContaining({ type: "text", text: "Hello, trunc" })]); }); + test("promotes max-output incomplete function calls with strict-complete arguments", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_complete", + call_id: "call_complete", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_complete", + delta: '{"path":"complete.txt"} \n\t', + }, + { + type: "response.incomplete", + response: { + id: "resp_complete_call", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("toolUse"); + expect(output.content).toHaveLength(1); + const block = output.content[0]; + if (block?.type !== "toolCall") throw new Error("expected a toolCall block"); + expect(block.arguments).toEqual({ path: "complete.txt" }); + }); + + test("keeps max-output incomplete function calls at length when only output_item.done closes arguments", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_closed", + call_id: "call_closed", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_closed", + delta: '{"path":"closed.txt"}', + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "function_call", + id: "fc_closed", + call_id: "call_closed", + name: "read", + arguments: '{"path":"closed.txt"}', + }, + }, + { + type: "response.incomplete", + response: { + id: "resp_closed_call", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + expect(output.content).toEqual([ + expect.objectContaining({ type: "toolCall", arguments: { path: "closed.txt" } }), + ]); + }); + + test("promotes max-output incomplete custom tool calls closed by input done", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + const patch = "*** Begin Patch\n*** End Patch"; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "custom_tool_call", + id: "ctc_closed", + call_id: "call_custom_closed", + name: "apply_patch", + input: "", + }, + }, + { + type: "response.custom_tool_call_input.delta", + output_index: 0, + item_id: "ctc_closed", + delta: patch, + }, + { + type: "response.custom_tool_call_input.done", + output_index: 0, + item_id: "ctc_closed", + input: patch, + }, + { + type: "response.incomplete", + response: { + id: "resp_closed_custom", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("toolUse"); + expect(output.content).toEqual([ + expect.objectContaining({ type: "toolCall", customWireName: "apply_patch", arguments: { input: patch } }), + ]); + }); + + test("keeps max-output incomplete custom tools at length when only output_item.done closes input", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + const patch = "*** Begin Patch\n*** End Patch"; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "custom_tool_call", + id: "ctc_output_done", + call_id: "call_custom_output_done", + name: "apply_patch", + input: "", + }, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "custom_tool_call", + id: "ctc_output_done", + call_id: "call_custom_output_done", + name: "apply_patch", + input: patch, + }, + }, + { + type: "response.incomplete", + response: { + id: "resp_output_done_custom", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + expect(output.content).toEqual([ + expect.objectContaining({ type: "toolCall", customWireName: "apply_patch", arguments: { input: patch } }), + ]); + }); + + test("keeps max-output incomplete turns at length when any function call has a JSON prefix", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_first", + call_id: "call_first", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_first", + delta: '{"path":"complete.txt"}', + }, + { + type: "response.output_item.added", + output_index: 1, + item: { + type: "function_call", + id: "fc_second", + call_id: "call_second", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 1, + item_id: "fc_second", + delta: '{"path":"truncated.txt"', + }, + { + type: "response.incomplete", + response: { + id: "resp_truncated_call", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + }); + + test("keeps max-output incomplete unfinished custom-tool input at length", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_with_custom", + call_id: "call_with_custom", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_with_custom", + delta: '{"path":"complete.txt"}', + }, + { + type: "response.output_item.added", + output_index: 1, + item: { + type: "custom_tool_call", + id: "ctc_unfinished", + call_id: "call_unfinished", + name: "apply_patch", + input: "", + }, + }, + { + type: "response.custom_tool_call_input.delta", + output_index: 1, + item_id: "ctc_unfinished", + delta: "*** Begin Patch", + }, + { + type: "response.incomplete", + response: { + id: "resp_unfinished_custom", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + const customCall = output.content.find(block => block.type === "toolCall" && block.customWireName !== undefined); + expect(customCall).toEqual( + expect.objectContaining({ + type: "toolCall", + customWireName: "apply_patch", + arguments: { input: "*** Begin Patch" }, + }), + ); + }); + for (const testCase of [ { name: "absent terminal content preserves streamed text",