fix(ai): execute complete tools at output limit

This commit is contained in:
usr_bin_roygbiv
2026-07-22 15:04:52 -05:00
parent e06ac0b787
commit 72cc72c74e
3 changed files with 367 additions and 6 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed OpenAI Responses token-cap truncations suppressing fully streamed function and custom tool calls whose inputs are complete.
## [17.0.6] - 2026-07-20
### Fixed
+41 -3
View File
@@ -2483,6 +2483,7 @@ export async function processResponsesStream<TApi extends Api>(
const entry = lookupOpenToolCallAlias(event, "custom_tool_call");
if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") {
finalizeCustomToolCallInputDone(entry.block, event.input);
entry.block[kStreamingArgumentsDone] = true;
}
} else if (event.type === "response.output_item.done") {
const item = structuredCloneJSON(event.item);
@@ -2602,6 +2603,10 @@ export async function processResponsesStream<TApi extends Api>(
}
} else if (terminalEvent) {
const response = terminalEvent.response;
const shouldPromoteIncompleteToolUse =
response?.status === "incomplete" &&
response.incomplete_details?.reason === "max_output_tokens" &&
hasExecutableIncompleteResponsesToolCalls(output);
finalizePendingResponsesToolCalls(output);
if (response?.id) {
output.responseId = response.id;
@@ -2638,7 +2643,11 @@ export async function processResponsesStream<TApi extends Api>(
kind: "content-blocked",
});
}
promoteResponsesToolUseStopReason(output, (response as { end_turn?: boolean } | undefined)?.end_turn);
promoteResponsesToolUseStopReason(
output,
(response as { end_turn?: boolean } | undefined)?.end_turn,
shouldPromoteIncompleteToolUse,
);
options?.onCompleted?.();
// `response.completed`/`response.incomplete`/`response.done` is the last event of a
// Responses stream. Stop pulling instead of waiting for the server to
@@ -2694,6 +2703,28 @@ export function mapOpenAIResponsesStopReason(status: ResponseStatus | undefined)
}
}
function hasExecutableIncompleteResponsesToolCalls(output: AssistantMessage): boolean {
let hasToolCall = false;
for (const block of output.content) {
if (block.type !== "toolCall") continue;
hasToolCall = true;
const pending = block as ToolCall & {
[kStreamingPartialJson]?: string;
[kStreamingArgumentsDone]?: boolean;
};
const rawArguments = pending[kStreamingPartialJson];
// `output_item.done` is not positive completion proof: our Responses
// compatibility encoder force-closes still-open calls before forwarding an
// upstream `length` stop. Only an explicit arguments/input-done event sets
// this marker; an open ordinary call can instead prove completion with its
// retained strict-complete JSON.
if (pending[kStreamingArgumentsDone]) continue;
if (pending.customWireName !== undefined || rawArguments === undefined) return false;
if (classifyJsonPrefix(rawArguments) !== "complete") return false;
}
return hasToolCall;
}
/**
* Finalize any streamed toolCall block whose `output_item.done` never arrived
* (lossy proxy, or a terminal event that raced the per-item done): parse the
@@ -2727,8 +2758,15 @@ export function finalizePendingResponsesToolCalls(output: AssistantMessage): voi
* re-samples instead of ending. Callers set `output.stopReason` from the wire
* status first via {@link mapOpenAIResponsesStopReason}.
*/
export function promoteResponsesToolUseStopReason(output: AssistantMessage, endTurn: boolean | undefined): void {
if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
export function promoteResponsesToolUseStopReason(
output: AssistantMessage,
endTurn: boolean | undefined,
promoteIncompleteToolUse = false,
): void {
if (
output.content.some(block => block.type === "toolCall") &&
(output.stopReason === "stop" || (promoteIncompleteToolUse && output.stopReason === "length"))
) {
output.stopReason = "toolUse";
}
if (endTurn === false && output.stopReason === "stop") {
@@ -1,9 +1,9 @@
// Terminal-event contracts for `processResponsesStream`:
//
// 1. `response.incomplete` is a terminal frame (max_output_tokens / content
// filter truncation). It must populate usage and map to stopReason
// "length" — previously it was ignored entirely, so truncated responses
// reported stopReason "stop" with zero usage and no cost.
// filter truncation). It must populate usage and normally map to stopReason
// "length", while a fully streamed function call remains executable as
// "toolUse".
// 2. `response.output_item.done` for a custom_tool_call must persist the final
// input on the stored content block and drop the transient `partialJson`
// accumulation buffer, mirroring the function_call branch.
@@ -120,6 +120,325 @@ describe("processResponsesStream: terminal events", () => {
expect(output.content).toEqual([expect.objectContaining({ type: "text", text: "Hello, trunc" })]);
});
test("promotes max-output incomplete function calls with strict-complete arguments", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
await processResponsesStream(
makeStream([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "function_call",
id: "fc_complete",
call_id: "call_complete",
name: "read",
arguments: "",
},
},
{
type: "response.function_call_arguments.delta",
output_index: 0,
item_id: "fc_complete",
delta: '{"path":"complete.txt"} \n\t',
},
{
type: "response.incomplete",
response: {
id: "resp_complete_call",
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
},
},
]),
output,
stream,
makeModel(),
);
expect(output.stopReason).toBe("toolUse");
expect(output.content).toHaveLength(1);
const block = output.content[0];
if (block?.type !== "toolCall") throw new Error("expected a toolCall block");
expect(block.arguments).toEqual({ path: "complete.txt" });
});
test("keeps max-output incomplete function calls at length when only output_item.done closes arguments", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
await processResponsesStream(
makeStream([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "function_call",
id: "fc_closed",
call_id: "call_closed",
name: "read",
arguments: "",
},
},
{
type: "response.function_call_arguments.delta",
output_index: 0,
item_id: "fc_closed",
delta: '{"path":"closed.txt"}',
},
{
type: "response.output_item.done",
output_index: 0,
item: {
type: "function_call",
id: "fc_closed",
call_id: "call_closed",
name: "read",
arguments: '{"path":"closed.txt"}',
},
},
{
type: "response.incomplete",
response: {
id: "resp_closed_call",
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
},
},
]),
output,
stream,
makeModel(),
);
expect(output.stopReason).toBe("length");
expect(output.content).toEqual([
expect.objectContaining({ type: "toolCall", arguments: { path: "closed.txt" } }),
]);
});
test("promotes max-output incomplete custom tool calls closed by input done", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
const patch = "*** Begin Patch\n*** End Patch";
await processResponsesStream(
makeStream([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "custom_tool_call",
id: "ctc_closed",
call_id: "call_custom_closed",
name: "apply_patch",
input: "",
},
},
{
type: "response.custom_tool_call_input.delta",
output_index: 0,
item_id: "ctc_closed",
delta: patch,
},
{
type: "response.custom_tool_call_input.done",
output_index: 0,
item_id: "ctc_closed",
input: patch,
},
{
type: "response.incomplete",
response: {
id: "resp_closed_custom",
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
},
},
]),
output,
stream,
makeModel(),
);
expect(output.stopReason).toBe("toolUse");
expect(output.content).toEqual([
expect.objectContaining({ type: "toolCall", customWireName: "apply_patch", arguments: { input: patch } }),
]);
});
test("keeps max-output incomplete custom tools at length when only output_item.done closes input", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
const patch = "*** Begin Patch\n*** End Patch";
await processResponsesStream(
makeStream([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "custom_tool_call",
id: "ctc_output_done",
call_id: "call_custom_output_done",
name: "apply_patch",
input: "",
},
},
{
type: "response.output_item.done",
output_index: 0,
item: {
type: "custom_tool_call",
id: "ctc_output_done",
call_id: "call_custom_output_done",
name: "apply_patch",
input: patch,
},
},
{
type: "response.incomplete",
response: {
id: "resp_output_done_custom",
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
},
},
]),
output,
stream,
makeModel(),
);
expect(output.stopReason).toBe("length");
expect(output.content).toEqual([
expect.objectContaining({ type: "toolCall", customWireName: "apply_patch", arguments: { input: patch } }),
]);
});
test("keeps max-output incomplete turns at length when any function call has a JSON prefix", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
await processResponsesStream(
makeStream([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "function_call",
id: "fc_first",
call_id: "call_first",
name: "read",
arguments: "",
},
},
{
type: "response.function_call_arguments.delta",
output_index: 0,
item_id: "fc_first",
delta: '{"path":"complete.txt"}',
},
{
type: "response.output_item.added",
output_index: 1,
item: {
type: "function_call",
id: "fc_second",
call_id: "call_second",
name: "read",
arguments: "",
},
},
{
type: "response.function_call_arguments.delta",
output_index: 1,
item_id: "fc_second",
delta: '{"path":"truncated.txt"',
},
{
type: "response.incomplete",
response: {
id: "resp_truncated_call",
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
},
},
]),
output,
stream,
makeModel(),
);
expect(output.stopReason).toBe("length");
});
test("keeps max-output incomplete unfinished custom-tool input at length", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
await processResponsesStream(
makeStream([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "function_call",
id: "fc_with_custom",
call_id: "call_with_custom",
name: "read",
arguments: "",
},
},
{
type: "response.function_call_arguments.delta",
output_index: 0,
item_id: "fc_with_custom",
delta: '{"path":"complete.txt"}',
},
{
type: "response.output_item.added",
output_index: 1,
item: {
type: "custom_tool_call",
id: "ctc_unfinished",
call_id: "call_unfinished",
name: "apply_patch",
input: "",
},
},
{
type: "response.custom_tool_call_input.delta",
output_index: 1,
item_id: "ctc_unfinished",
delta: "*** Begin Patch",
},
{
type: "response.incomplete",
response: {
id: "resp_unfinished_custom",
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
},
},
]),
output,
stream,
makeModel(),
);
expect(output.stopReason).toBe("length");
const customCall = output.content.find(block => block.type === "toolCall" && block.customWireName !== undefined);
expect(customCall).toEqual(
expect.objectContaining({
type: "toolCall",
customWireName: "apply_patch",
arguments: { input: "*** Begin Patch" },
}),
);
});
for (const testCase of [
{
name: "absent terminal content preserves streamed text",