diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 2635cccb0..df71674be 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -15,6 +15,7 @@ - Fixed hide-secrets placeholders conflicting with hashline edit headers by replacing hash-delimited tokens with the unambiguous `$$HASH$$` format ([#6631](https://github.com/can1357/oh-my-pi/issues/6631)). - Fixed the Docker `natives-builder` stage failing to build releases ≥ 17.1.1: the native audio stack added bindgen (miniaudio needs libclang) and a bundled-opus CMake build (needs cmake + make), none of which were installed in the slim builder image. - Fixed a configured `modelRoles.default` naming an extension-registered model (listed in `enabledModels`) silently running on a different in-scope provider's model. The startup model scope is resolved before extensions call `registerProvider()`, so the default role dropped out of scope and `buildSessionOptions` pinned `options.model` to the first scoped model — which marked the model "explicit" and suppressed the post-extension default-role re-resolution. A configured default that can't be found in the startup scope is now deferred so it re-resolves against the fully registered, still `enabledModels`-scoped catalog once extensions load ([#6694](https://github.com/can1357/oh-my-pi/issues/6694)). +- Fixed Parakeet speech-to-text failing to load `sherpa-onnx-node` from Windows source workspaces when Bun installed the wrapper under `packages/coding-agent/node_modules` but hoisted its native platform package to the repository root ([#6690](https://github.com/can1357/oh-my-pi/issues/6690)). - Fixed `omp usage` duplicating org-less legacy accounts as "no usage data" rows whenever any sibling report carried an organization (mixed pools of pre-org-capture rows and fresh org-scoped logins): an org-less account is now covered by its own org-less report, while org-attributed sibling reports still never count as its coverage. - `omp usage` revalidates the broker credential snapshot before rendering: live usage reports were previously paired with a disk-cached account list up to an hour old, so a just-completed re-login (org-less row upserted to org-scoped) rendered as a phantom duplicate until the cache expired. - Fixed Advisor requests reaching Anthropic-compatible endpoints without a provider-facing session identity: the separately constructed advisor `Agent` never had a metadata resolver installed, so its outbound requests omitted the `metadata.user_id` session id that the main and subagent agents carry. Each advisor now emits its own `advisorProviderSessionId` via `metadata.user_id`, resolved live so a token refresh surfaces the current `account_uuid`, giving Main, subagent, and Advisor traffic distinct, stable session ids for proxy routing and attribution ([#6625](https://github.com/can1357/oh-my-pi/issues/6625)). diff --git a/packages/coding-agent/src/stt/asr-worker.ts b/packages/coding-agent/src/stt/asr-worker.ts index 81c1adc22..a120bd12f 100644 --- a/packages/coding-agent/src/stt/asr-worker.ts +++ b/packages/coding-agent/src/stt/asr-worker.ts @@ -35,6 +35,7 @@ import { type SttModelKey, type TransformersSttModelSpec, } from "./models"; +import { loadSourceSherpaRuntime, type SherpaOfflineRecognizer, type SherpaRuntime } from "./sherpa-runtime"; const ASR_TASK = "automatic-speech-recognition"; const SHERPA_PACKAGE = "sherpa-onnx-node"; @@ -50,7 +51,6 @@ const HF_RESOLVE_BASE = "https://huggingface.co"; // Coalesce download progress so streaming a multi-hundred-MB model file doesn't // flood the IPC channel with one event per chunk. const PROGRESS_EMIT_BYTES = 4_000_000; -const sourceRequire = createRequire(import.meta.url); const sttModelDevicePreference = resolveTinyModelDevicePreference(); const sttModelDtypeOverride = resolveTinyModelDtypeOverride(); @@ -90,41 +90,6 @@ interface TransformersRuntime { ) => Promise; } -/** Recognition result returned by `sherpa-onnx-node`'s offline recognizer. */ -interface SherpaOfflineResult { - text?: string; -} - -/** A sherpa-onnx offline stream that accepts a single waveform before decoding. */ -interface SherpaOfflineStream { - acceptWaveform(audio: { samples: Float32Array; sampleRate: number }): void; -} - -interface SherpaOfflineRecognizer { - createStream(): SherpaOfflineStream; - decodeAsync(stream: SherpaOfflineStream): Promise; -} - -/** Offline recognizer config passed to `sherpa-onnx-node` (transducer family). */ -interface SherpaOfflineConfig { - modelConfig: { - transducer: { encoder: string; decoder: string; joiner: string }; - tokens: string; - modelType: string; - numThreads: number; - provider: string; - debug: number; - }; - decodingMethod: string; -} - -/** Subset of the native `sherpa-onnx-node` module surface we use. */ -interface SherpaRuntime { - OfflineRecognizer: { - createAsync(config: SherpaOfflineConfig): Promise; - }; -} - /** A warm model plus the engine that loaded it; cached per tier key. */ type LoadedModel = | { engine: "transformers"; pipeline: AutomaticSpeechRecognitionPipeline } @@ -182,7 +147,7 @@ function getSherpaRuntimeDir(): string { */ function loadSherpaRuntime(transport: SttTransport, requestId: string, modelKey: SttModelKey): Promise { return sherpaRuntime.load(async () => { - if (!isCompiledBinary()) return sourceRequire(SHERPA_PACKAGE) as SherpaRuntime; + if (!isCompiledBinary()) return loadSourceSherpaRuntime(import.meta.url); const runtimeDir = await ensureRuntimeInstalled({ runtimeDir: getSherpaRuntimeDir(), install: { dependencies: { [SHERPA_PACKAGE]: getSherpaVersionSpec() } }, diff --git a/packages/coding-agent/src/stt/sherpa-runtime.ts b/packages/coding-agent/src/stt/sherpa-runtime.ts new file mode 100644 index 000000000..c2ce2fb5a --- /dev/null +++ b/packages/coding-agent/src/stt/sherpa-runtime.ts @@ -0,0 +1,71 @@ +import { createRequire } from "node:module"; +import * as os from "node:os"; +import { resolveRuntimeModule } from "@oh-my-pi/pi-utils"; + +const SHERPA_PACKAGE = "sherpa-onnx-node"; + +interface SherpaOfflineResult { + text?: string; +} + +interface SherpaOfflineStream { + acceptWaveform(audio: { samples: Float32Array; sampleRate: number }): void; +} + +interface SherpaOfflineConfig { + modelConfig: { + transducer: { encoder: string; decoder: string; joiner: string }; + tokens: string; + modelType: string; + numThreads: number; + provider: string; + debug: number; + }; + decodingMethod: string; +} + +/** A sherpa-onnx recognizer instance used by the STT worker. */ +export interface SherpaOfflineRecognizer { + createStream(): SherpaOfflineStream; + decodeAsync(stream: SherpaOfflineStream): Promise; +} + +/** The native sherpa-onnx module surface used by the STT worker. */ +export interface SherpaRuntime { + OfflineRecognizer: { + createAsync(config: SherpaOfflineConfig): Promise; + }; +} + +/** Loads the nearest working source-workspace sherpa wrapper, including hoisted fallbacks. */ +export function loadSourceSherpaRuntime(sourceUrl: string): SherpaRuntime { + const sourceRequire = createRequire(sourceUrl); + const nearestEntry = sourceRequire.resolve(SHERPA_PACKAGE); + try { + return createRequire(nearestEntry)(nearestEntry); + } catch (error) { + if (!(error instanceof Error && error.message.startsWith("Could not find sherpa-onnx-node. Tried"))) { + throw error; + } + const platform = os.platform(); + const platformPackage = `sherpa-onnx-${platform === "win32" ? "win" : platform}-${os.arch()}`; + for (const nodeModules of sourceRequire.resolve.paths(SHERPA_PACKAGE) ?? []) { + if (!resolveRuntimeModule(nodeModules, platformPackage)) continue; + const entry = resolveRuntimeModule(nodeModules, SHERPA_PACKAGE); + if (!entry || entry === nearestEntry) continue; + try { + return createRequire(entry)(entry); + } catch (candidateError) { + if ( + !( + candidateError instanceof Error && + candidateError.message.startsWith("Could not find sherpa-onnx-node. Tried") + ) + ) { + throw candidateError; + } + } + } + throw error; + } +} diff --git a/packages/coding-agent/test/stt-sherpa-runtime.test.ts b/packages/coding-agent/test/stt-sherpa-runtime.test.ts new file mode 100644 index 000000000..830685b4f --- /dev/null +++ b/packages/coding-agent/test/stt-sherpa-runtime.test.ts @@ -0,0 +1,71 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { loadSourceSherpaRuntime } from "@oh-my-pi/pi-coding-agent/stt/sherpa-runtime"; +import { removeWithRetries } from "@oh-my-pi/pi-utils"; + +const PLATFORM_PACKAGE = `sherpa-onnx-${os.platform() === "win32" ? "win" : os.platform()}-${os.arch()}`; + +async function writePackage(nodeModules: string, name: string, source: string): Promise { + const dir = path.join(nodeModules, name); + await Bun.write(path.join(dir, "package.json"), JSON.stringify({ name, main: "index.js" })); + await Bun.write(path.join(dir, "index.js"), source); +} + +describe("sherpa source runtime resolution", () => { + let tmp = ""; + + afterEach(async () => { + await removeWithRetries(tmp); + }); + + it("loads the wrapper colocated with a workspace-hoisted platform addon", async () => { + tmp = await fs.mkdtemp(path.join(os.tmpdir(), "omp-sherpa-source-")); + const rootNodeModules = path.join(tmp, "node_modules"); + const packageNodeModules = path.join(tmp, "packages", "coding-agent", "node_modules"); + await writePackage( + rootNodeModules, + "sherpa-onnx-node", + "module.exports = { OfflineRecognizer: { createAsync() {} } };\n", + ); + await writePackage(rootNodeModules, PLATFORM_PACKAGE, "module.exports = {};\n"); + await writePackage( + packageNodeModules, + "sherpa-onnx-node", + "throw new Error('Could not find sherpa-onnx-node. Tried');\n", + ); + + const sourceUrl = path.join(tmp, "packages", "coding-agent", "src", "stt", "asr-worker.ts"); + const runtime = loadSourceSherpaRuntime(sourceUrl); + + expect(runtime.OfflineRecognizer.createAsync).toBeTypeOf("function"); + }); + + it("prefers the nearest wrapper when its nested platform addon is loadable", async () => { + tmp = await fs.mkdtemp(path.join(os.tmpdir(), "omp-sherpa-source-")); + const rootNodeModules = path.join(tmp, "node_modules"); + const packageNodeModules = path.join(tmp, "packages", "coding-agent", "node_modules"); + await writePackage( + rootNodeModules, + "sherpa-onnx-node", + "module.exports = { OfflineRecognizer: { createAsync: function rootRuntime() {} } };\n", + ); + await writePackage(rootNodeModules, PLATFORM_PACKAGE, "module.exports = {};\n"); + await writePackage( + packageNodeModules, + "sherpa-onnx-node", + `require("./node_modules/${PLATFORM_PACKAGE}"); module.exports = { OfflineRecognizer: { createAsync: function nestedRuntime() {} } };\n`, + ); + await writePackage( + path.join(packageNodeModules, "sherpa-onnx-node", "node_modules"), + PLATFORM_PACKAGE, + "module.exports = {};\n", + ); + + const sourceUrl = path.join(tmp, "packages", "coding-agent", "src", "stt", "asr-worker.ts"); + const runtime = loadSourceSherpaRuntime(sourceUrl); + + expect(runtime.OfflineRecognizer.createAsync.name).toBe("nestedRuntime"); + }); +});