feat: added strict-mode fallback for OpenAI tool calls with all_strict

- Added `toolStrictMode` support with `all_strict`/`none`/`mixed` options to OpenAI compatibility.
- Fixed OpenAI-completion strict-mode flows by capturing failed HTTP responses and retrying once as non-strict.
- Fixed completion error reporting by surfacing captured status, headers, and JSON `type`/`param`/`code` details.
- Improved strict-schema enforcement with WeakMap memoization and circular-schema detection in sanitization.
- Fixed OpenRouter provider lookup by resolving fallback model IDs for suffix and date variants in registry resolution.
- Refactored benchmark tooling and added async RPC error-window tracking for scheduled run execution.
This commit is contained in:
can1357
2026-04-13 15:30:34 +02:00
parent b69dea4823
commit 212d56bc11
38 changed files with 1492 additions and 548 deletions
@@ -1132,7 +1132,6 @@ async function runSingleTask(
}
}
// Retry if the model didn't attempt any edit/write (read-only or no tool calls)
const madeEditAttempt = toolStats.edit > 0 || toolStats.write > 0;
if (!madeEditAttempt && zeroToolRetries < noOpRetryLimit) {
@@ -1413,7 +1412,7 @@ async function _runRpcBenchmarkRun(
} else if (toolName === "write") {
toolStats.write++;
}
if (e.args) {
toolStats.totalInputChars += JSON.stringify(e.args).length;
}
@@ -1460,7 +1459,7 @@ async function _runRpcBenchmarkRun(
}
}
}
// Retry if the model didn't attempt any edit/write (read-only or no tool calls)
const madeEditAttempt = toolStats.edit > 0 || toolStats.write > 0;
if (!madeEditAttempt && zeroToolRetries < noOpRetryLimit) {
@@ -1470,15 +1469,15 @@ async function _runRpcBenchmarkRun(
attempt--; // Don't consume a regular attempt slot
continue;
}
patchApplied = toolStats.edit > 0;
const filesToVerify = task.files.length > 0 ? task.files : undefined;
const verification = await verifyExpectedFileSubset(expectedDir, cwd, filesToVerify);
if (config.autoFormat) {
await formatDirectory(cwd);
}
verificationPassed = verification.success;
indentScore = verification.indentScore;
formattedEquivalent = verification.formattedEquivalent;
@@ -1488,11 +1487,11 @@ async function _runRpcBenchmarkRun(
if (!verification.success && verification.error) {
error = verification.error;
}
if (verification.success) {
break;
}
const mutationIntentSuffix = mutationIntentValidation
? `\n\nMutation intent: ${mutationIntentValidation.matched ? "matched" : "not matched"} (${mutationIntentValidation.reason})`
: "";