feat(coding-agent): added provider extension support to CLI tools

- Enable shared extension provider loading in bench and dry-balance CLI commands to ensure custom providers are registered.
- Surface benchmark failures for empty streams that return no content and zero usage tokens instead of treating them as successful.
This commit is contained in:
can1357
2026-06-19 19:23:46 +02:00
parent eb784d7ae6
commit d10a5356ca
3 changed files with 68 additions and 4 deletions
+33 -2
View File
@@ -25,7 +25,7 @@ import {
} from "../config/model-resolver";
import { Settings } from "../config/settings";
import benchPrompt from "../prompts/bench.md" with { type: "text" };
import { discoverAuthStorage } from "../sdk";
import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk";
import { resolveThinkingLevelForModel, shouldDisableReasoning, toReasoningEffort } from "../thinking";
const DEFAULT_RUNS = 1;
@@ -145,6 +145,23 @@ function isFirstTokenEvent(event: AssistantMessageEvent): boolean {
}
}
/** Final message carries visible output — non-empty text/thinking or a tool call. */
function hasVisibleFinalContent(message: AssistantMessage): boolean {
return message.content.some(block => {
switch (block.type) {
case "text":
return block.text.length > 0;
case "thinking":
return block.thinking.length > 0;
case "redactedThinking":
case "toolCall":
return true;
default:
return false;
}
});
}
/**
* Tokens/s over the generation window (duration minus TTFT) so queue/prefill
* latency does not dilute throughput. Falls back to total duration when the
@@ -232,6 +249,18 @@ async function runBenchRequest(
const rawTtft = message.ttft ?? (firstTokenAt === undefined ? durationMs : firstTokenAt - startedAt);
const ttftMs = Number.isFinite(rawTtft) && rawTtft > 0 ? rawTtft : 0;
const outputTokens = Number.isFinite(message.usage.output) && message.usage.output > 0 ? message.usage.output : 0;
// A run that streamed no content (no delta/end event set firstTokenAt),
// carries no visible final content, and measured no output tokens
// benchmarked nothing — a genuinely empty stream (e.g. a gateway that 200s
// with an empty body). Surface it as a failure instead of a misleading
// 0-token "✓". Streaming and buffered providers that produce content keep
// passing even when usage is omitted.
if (firstTokenAt === undefined && outputTokens === 0 && !hasVisibleFinalContent(message)) {
return {
ok: false,
error: `provider returned no output (0 tokens, empty stream; stop reason: ${message.stopReason ?? "unknown"})`,
};
}
return {
ok: true,
ttftMs,
@@ -328,8 +357,10 @@ export function formatBenchTable(summary: BenchSummary): string {
async function createDefaultRuntime(): Promise<BenchRuntime> {
const authStorage = await discoverAuthStorage();
try {
const settings = await Settings.init({ cwd: getProjectDir() });
const cwd = getProjectDir();
const settings = await Settings.init({ cwd });
const modelRegistry = new ModelRegistry(authStorage);
await loadCliExtensionProviders(modelRegistry, settings, cwd);
return {
modelRegistry,
settings,
@@ -26,7 +26,7 @@ import {
} from "../config/model-resolver";
import { Settings } from "../config/settings";
import dryBalanceBenchPrompt from "../prompts/dry-balance-bench.md" with { type: "text" };
import { discoverAuthStorage } from "../sdk";
import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk";
const DEFAULT_SAMPLE_COUNT = 100;
const DEFAULT_CONCURRENCY = 32;
@@ -523,8 +523,10 @@ async function runBenchTargets(
async function createDefaultRuntime(): Promise<DryBalanceRuntime> {
const authStorage = await discoverAuthStorage();
try {
const settings = await Settings.init({ cwd: getProjectDir() });
const cwd = getProjectDir();
const settings = await Settings.init({ cwd });
const modelRegistry = new ModelRegistry(authStorage);
await loadCliExtensionProviders(modelRegistry, settings, cwd);
return {
modelRegistry,
settings,
+31
View File
@@ -687,6 +687,37 @@ export async function loadSessionExtensions(
return result;
}
/**
* Load discovered/configured extensions and register their providers into
* `modelRegistry`, then discover the dynamic provider catalogs. One-shot CLIs
* (`omp bench`, dry-balance) build a bare {@link ModelRegistry} that only knows
* built-in catalog providers; without this, providers contributed by an
* extension (e.g. a custom OpenAI-compatible provider under
* `~/.omp/agent/extensions/`) never reach model resolution. Mirrors the
* session / `omp models` path: drain the queued provider registrations, then
* `refreshRuntimeProviders` so dynamically-discovered models exist before
* selectors are resolved.
*/
export async function loadCliExtensionProviders(
modelRegistry: ModelRegistry,
settings: Settings,
cwd: string,
options: Pick<CreateAgentSessionOptions, "disableExtensionDiscovery" | "additionalExtensionPaths"> = {},
): Promise<void> {
const eventBus = new EventBus();
const extensionsResult = await loadSessionExtensions(options, cwd, settings, eventBus);
const activeSources = extensionsResult.extensions.map(extension => extension.path);
modelRegistry.syncExtensionSources(activeSources);
for (const sourceId of new Set(activeSources)) {
modelRegistry.clearSourceRegistrations(sourceId);
}
for (const { name, config, sourceId } of extensionsResult.runtime.pendingProviderRegistrations) {
modelRegistry.registerProvider(name, config, sourceId);
}
extensionsResult.runtime.pendingProviderRegistrations = [];
await modelRegistry.refreshRuntimeProviders();
}
/**
* Discover skills from cwd and agentDir.
*/