feat(ai): retried strict tool requests when structured outputs are unsupported
- Added a fallback mechanism to automatically downgrade from strict to non-strict tool mode when gateways report "structured_outputs not supported". - Updated error classification to recognize structured output rejections from feature-gated deployments like Azure Foundry. - Updated Anthropic and OpenAI providers to handle these 400-level rejections and persist the non-strict setting for the duration of the session.
This commit is contained in:
@@ -5,6 +5,7 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama/Ollama Cloud EOS-only completions to retry empty stops with a single output token before the agent loop can halt silently. ([#4659](https://github.com/can1357/oh-my-pi/issues/4659))
|
||||
- Fixed Claude Sonnet 5 failing every request on feature-gated gateways (Azure Foundry, OpenAI-compatible relays) that reject strict tools with "structured_outputs not supported" — the rejection is now classified as a strict-tool rejection, so the request retries without strict tools and the session remembers the downgrade.
|
||||
|
||||
## [16.3.7] - 2026-07-05
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ export const Flag = {
|
||||
SilentAbort: 0x0200_0000,
|
||||
UserInterrupt: 0x0400_0000,
|
||||
Abort: 0x0800_0000,
|
||||
/** Anthropic strict-tool grammar too large / schema too complex to compile (400). */
|
||||
/** Strict-tool rejection (400): grammar too large, schema too complex, or structured outputs unsupported by the model/endpoint. */
|
||||
Grammar: 0x1000_0000,
|
||||
/** Anthropic model/account does not support fast mode / the `speed` parameter. */
|
||||
FastModeUnsupported: 0x2000_0000,
|
||||
@@ -108,12 +108,17 @@ export const LLAMA_CPP_TOOL_CALL_PARSE_PATTERN =
|
||||
// on a backend that has the model.
|
||||
const COPILOT_MODEL_NOT_SUPPORTED_PATTERN = /model_not_supported/i;
|
||||
// Anthropic strict-tool grammar too large / schema too complex (400 invalid_request_error).
|
||||
// Feature-gated deployments (Azure Foundry, Baseten, …) reject `strict: true`
|
||||
// tools outright when the hosted model lacks structured outputs, e.g.
|
||||
// "structured_outputs not supported" — without an invalid_request_error wrapper.
|
||||
const GRAMMAR_TOO_LARGE_PATTERN = /compiled grammar/i;
|
||||
const GRAMMAR_TOO_LARGE_DETAIL_PATTERN = /too large/i;
|
||||
const SCHEMA_TOO_COMPLEX_PATTERN = /schema/i;
|
||||
const SCHEMA_TOO_COMPLEX_DETAIL_PATTERN = /too complex/i;
|
||||
const SCHEMA_COMPILE_PATTERN = /compil/i;
|
||||
const INVALID_REQUEST_PATTERN = /invalid_request_error/i;
|
||||
const STRUCTURED_OUTPUTS_PATTERN = /structured[_ -]?outputs?/i;
|
||||
const FEATURE_NOT_SUPPORTED_PATTERN = /not (?:supported|available|enabled)|unsupported|does(?: not|n'?t) support/i;
|
||||
// Anthropic fast-mode unsupported: 400 rejecting `speed`, or 429 rate_limit_error
|
||||
// because the account lacks the extra-usage entitlement fast mode requires.
|
||||
const FAST_MODE_SPEED_PARAM_PATTERN = /\bspeed\b/i;
|
||||
@@ -127,8 +132,9 @@ const OAUTH_TRANSIENT_FAILURE_PATTERN =
|
||||
/timeout|network|fetch failed|ECONN(?:REFUSED|RESET)|ETIMEDOUT|EAI_AGAIN|socket hang up|\b(?:408|425|429|5\d{2})\b|rate.?limit|too many requests|temporar|unavailable|forbidden|permission_denied|cloudflare|captcha/i;
|
||||
const OAUTH_HTTP_AUTH_PATTERN = /\b401\b/;
|
||||
|
||||
function matchesGrammarTooLarge(message: string, errorStatus: number | undefined): boolean {
|
||||
function matchesStrictToolsRejection(message: string, errorStatus: number | undefined): boolean {
|
||||
if (errorStatus !== 400) return false;
|
||||
if (STRUCTURED_OUTPUTS_PATTERN.test(message) && FEATURE_NOT_SUPPORTED_PATTERN.test(message)) return true;
|
||||
if (!INVALID_REQUEST_PATTERN.test(message)) return false;
|
||||
const grammarTooLarge = GRAMMAR_TOO_LARGE_PATTERN.test(message) && GRAMMAR_TOO_LARGE_DETAIL_PATTERN.test(message);
|
||||
const schemaTooComplex =
|
||||
@@ -317,7 +323,7 @@ function classifyText(errorMessage: string | undefined, errorStatus: number | un
|
||||
|
||||
// Copilot per-client routing flap is transient.
|
||||
if (statusClean === 400 && COPILOT_MODEL_NOT_SUPPORTED_PATTERN.test(cleanMessage)) kinds |= Flag.Transient;
|
||||
if (matchesGrammarTooLarge(cleanMessage, statusClean)) kinds |= Flag.Grammar;
|
||||
if (matchesStrictToolsRejection(cleanMessage, statusClean)) kinds |= Flag.Grammar;
|
||||
if (matchesFastModeUnsupported(cleanMessage, statusClean)) kinds |= Flag.FastModeUnsupported;
|
||||
}
|
||||
if (kinds !== 0) return create(kinds);
|
||||
@@ -411,7 +417,8 @@ export function isUsageLimit(error: unknown, api?: Api): boolean {
|
||||
}
|
||||
|
||||
/**
|
||||
* Anthropic strict-tool grammar too large / schema too complex to compile.
|
||||
* Strict-tool rejection: grammar too large, schema too complex, or structured
|
||||
* outputs unsupported by the model/endpoint.
|
||||
* Accessor for {@link Flag.Grammar}.
|
||||
*/
|
||||
export function isGrammarError(error: unknown): boolean {
|
||||
|
||||
@@ -2396,7 +2396,7 @@ const streamAnthropicOnce = (
|
||||
) {
|
||||
// Log-only: the retried turn must not carry an errorMessage on
|
||||
// success (consumers treat its presence as failure).
|
||||
logger.warn("anthropic: strict tool grammar rejected, retrying without strict tools", {
|
||||
logger.warn("anthropic: strict tools rejected, retrying without strict tools", {
|
||||
model: model.id,
|
||||
error: await finalizeErrorMessage(streamFailure, rawRequestDump),
|
||||
});
|
||||
|
||||
@@ -1041,7 +1041,7 @@ export function shouldRetryWithoutStrictTools(
|
||||
const messageParts = [error instanceof Error ? error.message : undefined, capturedErrorResponse?.bodyText]
|
||||
.filter((value): value is string => typeof value === "string" && value.trim().length > 0)
|
||||
.join("\n");
|
||||
return /wrong_api_format|mixed values for 'strict'|tool[s]?\b.*strict|\bstrict\b.*tool|tool parameters? schema|invalid schema for function/i.test(
|
||||
return /wrong_api_format|mixed values for 'strict'|tool[s]?\b.*strict|\bstrict\b.*tool|tool parameters? schema|invalid schema for function|structured[_ -]?outputs?\b[^\n]*(?:not (?:supported|available|enabled)|unsupported)|(?:not support|unsupported)[^\n]*structured[_ -]?outputs?\b/i.test(
|
||||
messageParts,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -139,6 +139,14 @@ function createStrictGrammarTooLargeError(): Error {
|
||||
return error;
|
||||
}
|
||||
|
||||
// Azure Foundry-style rejection: no invalid_request_error wrapper, the gateway
|
||||
// just names the missing feature for the hosted model deployment.
|
||||
function createStructuredOutputsUnsupportedError(): Error {
|
||||
const error = new Error('400 {"error":{"code":"BadRequest","message":"structured_outputs not supported"}}');
|
||||
(error as Error & { status: number }).status = 400;
|
||||
return error;
|
||||
}
|
||||
|
||||
function createOtherInvalidRequestError(): Error {
|
||||
const error = new Error(
|
||||
'400 {"type":"error","error":{"type":"invalid_request_error","message":"Some other validation error."},"request_id":"req_test"}',
|
||||
@@ -1015,6 +1023,45 @@ describe("anthropic stream envelope handling", () => {
|
||||
expect(strictFlags).toEqual([[true], [false], [false]]);
|
||||
});
|
||||
|
||||
it("retries without strict tools when the endpoint rejects structured outputs for the model", async () => {
|
||||
const toolContext: Context = {
|
||||
...context,
|
||||
tools: [
|
||||
{
|
||||
name: "edit",
|
||||
description: "Edit a value",
|
||||
strict: true,
|
||||
parameters: queryObjectSchema,
|
||||
},
|
||||
],
|
||||
};
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const strictFlags: boolean[][] = [];
|
||||
let attempt = 0;
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation((params: unknown) => {
|
||||
attempt += 1;
|
||||
strictFlags.push(getStrictFlags(params));
|
||||
if (attempt === 1) {
|
||||
return createRejectedMockRequest(createStructuredOutputsUnsupportedError()) as never;
|
||||
}
|
||||
return createMockRequest(createTextSuccessEvents("recovered")) as never;
|
||||
});
|
||||
|
||||
const stream = streamAnthropic(model, toolContext, { apiKey: "sk-ant-test", providerSessionState });
|
||||
const events: AssistantMessageEvent[] = [];
|
||||
for await (const event of stream) {
|
||||
events.push(event);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(JSON.parse(JSON.stringify(result.content))).toEqual([{ type: "text", text: "recovered" }]);
|
||||
expect(countEvents(events, "error")).toBe(0);
|
||||
expect(strictFlags).toEqual([[true], [false]]);
|
||||
expect(anthropicStrictToolsDisabled(providerSessionState)).toBe(true);
|
||||
});
|
||||
|
||||
it("does not disable strict tools for unrelated Anthropic invalid request errors", async () => {
|
||||
const toolContext: Context = {
|
||||
...context,
|
||||
|
||||
@@ -416,6 +416,58 @@ describe("OpenAI tool strict mode", () => {
|
||||
expect(strictFlags).toEqual([[true], [false], [false]]);
|
||||
});
|
||||
|
||||
it("retries non-strict when a gateway rejects structured outputs for the model", async () => {
|
||||
const model = getBundledModel("openrouter", "anthropic/claude-sonnet-4") as Model<"openai-completions">;
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const strictFlags: boolean[][] = [];
|
||||
let attempt = 0;
|
||||
const fetchMock: FetchImpl = Object.assign(
|
||||
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
attempt += 1;
|
||||
const bodyText = typeof init?.body === "string" ? init.body : "";
|
||||
const payload = JSON.parse(bodyText) as {
|
||||
tools?: Array<{ function?: { strict?: boolean } }>;
|
||||
};
|
||||
strictFlags.push((payload.tools ?? []).map(tool => tool.function?.strict === true));
|
||||
if (attempt === 1) {
|
||||
return new Response(
|
||||
JSON.stringify({ error: { code: "BadRequest", message: "structured_outputs not supported" } }),
|
||||
{ status: 400, headers: { "content-type": "application/json" } },
|
||||
);
|
||||
}
|
||||
return createSseResponse([
|
||||
{
|
||||
id: "chatcmpl-foundry-retry",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: model.id,
|
||||
choices: [{ index: 0, delta: { content: "Recovered" } }],
|
||||
},
|
||||
{
|
||||
id: "chatcmpl-foundry-retry",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: model.id,
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
},
|
||||
"[DONE]",
|
||||
]);
|
||||
},
|
||||
{ preconnect: fetch.preconnect },
|
||||
);
|
||||
|
||||
const result = await streamOpenAICompletions(model, testContext, {
|
||||
apiKey: "test-key",
|
||||
providerSessionState,
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.content).toContainEqual({ type: "text", text: "Recovered" });
|
||||
expect(strictFlags).toEqual([[true], [false]]);
|
||||
});
|
||||
|
||||
it("clears errorMessage on a successful OpenRouter Anthropic compiled-grammar fallback (responses)", async () => {
|
||||
const model = buildModel({
|
||||
id: "anthropic/claude-sonnet-4",
|
||||
|
||||
Reference in New Issue
Block a user