fix(tiny): preserved cpu fallback when cuda sidecar repair fails
Deferred ONNX Runtime CUDA sidecar repair failure into the runtime metadata so loadTransformersRuntime keeps loading and loadPipelineWithDeviceFallback still gets its CUDA→CPU retry when NuGet is offline or the ort install script is unavailable. The failure surfaces through the CUDA diagnostics helper instead of hard-erroring the tiny worker. Refs #4475
This commit is contained in:
@@ -300,6 +300,7 @@ interface ConfigurableTransformers {
|
||||
export interface TransformersRuntimeMetadata {
|
||||
__ompRuntimeNodeModules?: string;
|
||||
__ompTransformersEntry?: string;
|
||||
__ompCudaRepairError?: string;
|
||||
}
|
||||
|
||||
function attachTransformersRuntimeMetadata<T extends ConfigurableTransformers>(
|
||||
@@ -309,6 +310,7 @@ function attachTransformersRuntimeMetadata<T extends ConfigurableTransformers>(
|
||||
const runtime = transformers as T & TransformersRuntimeMetadata;
|
||||
runtime.__ompRuntimeNodeModules = metadata.__ompRuntimeNodeModules;
|
||||
runtime.__ompTransformersEntry = metadata.__ompTransformersEntry;
|
||||
runtime.__ompCudaRepairError = metadata.__ompCudaRepairError;
|
||||
return runtime;
|
||||
}
|
||||
|
||||
@@ -324,7 +326,14 @@ function missingCudaLibrary(error: unknown): string | undefined {
|
||||
return TRANSITIVE_CUDA_LIBRARY_RE.exec(errorText(error))?.[1];
|
||||
}
|
||||
|
||||
function cudaFailureCause(error: unknown, missingFiles: readonly string[]): string {
|
||||
function cudaFailureCause(
|
||||
metadata: TransformersRuntimeMetadata,
|
||||
error: unknown,
|
||||
missingFiles: readonly string[],
|
||||
): string {
|
||||
if (metadata.__ompCudaRepairError) {
|
||||
return `ONNX Runtime CUDA provider install failed: ${metadata.__ompCudaRepairError}`;
|
||||
}
|
||||
if (missingFiles.length > 0) return `missing ONNX Runtime CUDA provider file(s): ${missingFiles.join(", ")}`;
|
||||
const missingLibrary = missingCudaLibrary(error);
|
||||
if (missingLibrary) return `${missingLibrary}: cannot open shared object file`;
|
||||
@@ -334,7 +343,14 @@ function cudaFailureCause(error: unknown, missingFiles: readonly string[]): stri
|
||||
return "CUDA provider files are present; inspect the original ONNX Runtime CUDA error";
|
||||
}
|
||||
|
||||
function cudaFailureHint(error: unknown, missingFiles: readonly string[]): string {
|
||||
function cudaFailureHint(
|
||||
metadata: TransformersRuntimeMetadata,
|
||||
error: unknown,
|
||||
missingFiles: readonly string[],
|
||||
): string {
|
||||
if (metadata.__ompCudaRepairError) {
|
||||
return "restore network access to nuget.org (or pre-populate the tiny side runtime) and rerun; CPU inference remained available";
|
||||
}
|
||||
if (missingFiles.length > 0) return "reinstall the tiny side runtime with ONNX Runtime postinstall enabled";
|
||||
if (missingCudaLibrary(error)) {
|
||||
return "install the matching CUDA/cuDNN shared libraries and expose them on the dynamic loader path";
|
||||
@@ -383,9 +399,9 @@ export async function formatOnnxRuntimeCudaDiagnostics(
|
||||
"ONNX Runtime CUDA diagnostics:",
|
||||
` PI_TINY_DEVICE=${requestedDevice} requested CUDAExecutionProvider`,
|
||||
sideRuntime ? ` side runtime: ${sideRuntime}` : ` onnxruntime-node: ${packageDir}`,
|
||||
` cause: ${cudaFailureCause(error, missingFiles)}`,
|
||||
` cause: ${cudaFailureCause(metadata, error, missingFiles)}`,
|
||||
];
|
||||
lines.push(` hint: ${cudaFailureHint(error, missingFiles)}`);
|
||||
lines.push(` hint: ${cudaFailureHint(metadata, error, missingFiles)}`);
|
||||
return lines.join("\n");
|
||||
}
|
||||
|
||||
@@ -453,12 +469,21 @@ export function loadTransformersRuntime<T extends ConfigurableTransformers, K>(
|
||||
},
|
||||
}),
|
||||
});
|
||||
await ensureOnnxRuntimeCudaProviders(installedDir);
|
||||
let cudaRepairError: string | undefined;
|
||||
try {
|
||||
await ensureOnnxRuntimeCudaProviders(installedDir);
|
||||
} catch (repairError) {
|
||||
// Deferred failure: keep loading Transformers so `loadPipelineWithDeviceFallback`
|
||||
// still gets its CUDA→CPU retry. The error is surfaced through the CUDA
|
||||
// diagnostics attached to the runtime metadata.
|
||||
cudaRepairError = errorMessage(repairError);
|
||||
}
|
||||
const entry = await prepareCompiledRuntime(installedDir, TRANSFORMERS_PACKAGE);
|
||||
const require_ = createRequire(entry);
|
||||
return attachTransformersRuntimeMetadata(configureTransformers(require_(entry) as T), {
|
||||
__ompRuntimeNodeModules: path.join(installedDir, "node_modules"),
|
||||
__ompTransformersEntry: entry,
|
||||
__ompCudaRepairError: cudaRepairError,
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -67,4 +67,30 @@ describe("tiny runtime CUDA provider repair", () => {
|
||||
expect(diagnostic).toContain("CUDA runtime reports no CUDA-capable device");
|
||||
expect(diagnostic).toContain("make the NVIDIA GPU visible to this process/session");
|
||||
});
|
||||
|
||||
it("surfaces a deferred sidecar install failure through the CUDA diagnostics helper", async () => {
|
||||
if (process.platform !== "linux" || process.arch !== "x64") return;
|
||||
const runtimeDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-tiny-runtime-install-"));
|
||||
tempDirs.push(runtimeDir);
|
||||
const packageDir = path.join(runtimeDir, "node_modules", "onnxruntime-node");
|
||||
await Bun.write(
|
||||
path.join(packageDir, "package.json"),
|
||||
JSON.stringify({ name: "onnxruntime-node", version: "1.24.3", main: "dist/index.js" }),
|
||||
);
|
||||
await Bun.write(path.join(packageDir, "dist", "index.js"), "module.exports = {};\n");
|
||||
|
||||
const diagnostic = await formatOnnxRuntimeCudaDiagnostics(
|
||||
{
|
||||
__ompRuntimeNodeModules: path.join(runtimeDir, "node_modules"),
|
||||
__ompCudaRepairError:
|
||||
"Failed to install ONNX Runtime CUDA provider binaries: connect ENETUNREACH api.nuget.org",
|
||||
},
|
||||
"cuda",
|
||||
new Error("OrtSessionOptionsAppendExecutionProvider_Cuda: Failed to load shared library"),
|
||||
);
|
||||
|
||||
expect(diagnostic).toContain("ONNX Runtime CUDA provider install failed");
|
||||
expect(diagnostic).toContain("ENETUNREACH");
|
||||
expect(diagnostic).toContain("CPU inference remained available");
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user