diff --git a/gitnexus/scripts/cross-platform-tests.ts b/gitnexus/scripts/cross-platform-tests.ts index 514c52dcc..c4ec8d734 100644 --- a/gitnexus/scripts/cross-platform-tests.ts +++ b/gitnexus/scripts/cross-platform-tests.ts @@ -47,6 +47,7 @@ const PLATFORM_LOGIC = [ 'test/unit/ignore-service.test.ts', 'test/unit/group/bridge-db.test.ts', 'test/unit/group/bridge-db-edge.test.ts', + 'test/unit/onnxruntime-node-resolver.test.ts', ]; // Native LadybugDB integration tests — exercise the @ladybugdb/core diff --git a/gitnexus/src/cli/doctor.ts b/gitnexus/src/cli/doctor.ts index fb83d2e5c..adbb39801 100644 --- a/gitnexus/src/cli/doctor.ts +++ b/gitnexus/src/cli/doctor.ts @@ -2,6 +2,7 @@ import { getRuntimeCapabilities, getRuntimeFingerprint } from '../core/platform/ import { resolveEmbeddingConfig } from '../core/embeddings/config.js'; import { isHttpMode } from '../core/embeddings/http-client.js'; import { getLocalEmbeddingRuntimeBlocker } from '../core/embeddings/runtime-support.js'; +import { cudaRedirectDoctorStatus } from '../core/embeddings/onnxruntime-node-resolver.js'; import { checkLbugNative } from '../core/lbug/native-check.js'; import { getExtensionInstallPolicy } from '../core/lbug/extension-loader.js'; import { t } from './i18n/index.js'; @@ -139,4 +140,14 @@ export const doctorCommand = async () => { if (support.detail) { process.stderr.write(`\n${support.detail.replace(/^/gm, ' ')}\n\n`); } + // Surface the CUDA-build-redirect decision so "why is my CUDA-13 host + // still on CPU" is visible without digging through debug logs (#2341 + // follow-up). Only meaningful on the local runtime path. + if (!isHttpMode()) { + const cudaRedirect = cudaRedirectDoctorStatus(); + console.log(` ${padDisplayEnd('CUDA:', 12)}${cudaRedirect.status}`); + if (cudaRedirect.detail) { + console.log(` ${padDisplayEnd('', 12)}${cudaRedirect.detail}`); + } + } }; diff --git a/gitnexus/src/core/embeddings/embedder.ts b/gitnexus/src/core/embeddings/embedder.ts index 3ec5f08e1..800652a31 100644 --- a/gitnexus/src/core/embeddings/embedder.ts +++ b/gitnexus/src/core/embeddings/embedder.ts @@ -19,94 +19,18 @@ if (!process.env.ORT_LOG_LEVEL) { // runtime. The runtime values (pipeline, env) are dynamically imported inside // initEmbedder, after the platform guard has passed (#1515). import type { FeatureExtractionPipeline, ProgressInfo } from '@huggingface/transformers'; -import { existsSync } from 'fs'; -import { execFileSync } from 'child_process'; -import { join, dirname } from 'path'; -import { createRequire } from 'module'; import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types.js'; import { isHttpMode, getHttpDimensions, httpEmbed } from './http-client.js'; import { resolveEmbeddingConfig } from './config.js'; import { applyHfEnvOverrides, isHfDownloadFailure, withHfDownloadRetry } from './hf-env.js'; import { getLocalEmbeddingRuntimeBlocker } from './runtime-support.js'; import { ensureOnnxRuntimeCommonResolvable } from './onnxruntime-common-resolver.js'; +import { + ensureOnnxRuntimeNodeMatchesSystem, + isEffectiveCudaAvailable, +} from './onnxruntime-node-resolver.js'; import { logger } from '../logger.js'; -/** - * Check whether the onnxruntime-node package that @huggingface/transformers - * will actually load at runtime ships the CUDA execution provider. - * - * Critical: we resolve from transformers' own module scope, NOT from ours. - * npm may install two copies — a top-level 1.24.x (our dep) and a nested - * 1.21.0 (transformers' pinned dep). The guard must inspect whichever copy - * transformers.js will dlopen, otherwise the check is meaningless. - */ -function hasOrtCudaProvider(): boolean { - try { - const require = createRequire(import.meta.url); - // Resolve from @huggingface/transformers' scope so we find the same - // onnxruntime-node binary that transformers.js will use at runtime - const transformersDir = dirname(require.resolve('@huggingface/transformers/package.json')); - const ortRequire = createRequire(join(transformersDir, 'package.json')); - const ortPath = dirname(ortRequire.resolve('onnxruntime-node/package.json')); - // ORT 1.24.x only ships CUDA binaries for linux/x64 (downloaded from NuGet - // at postinstall). arm64 will correctly return false here until ORT adds support. - const arch = process.arch; - return existsSync( - join(ortPath, 'bin', 'napi-v6', 'linux', arch, 'libonnxruntime_providers_cuda.so'), - ); - } catch { - return false; - } -} - -/** - * Check whether CUDA libraries are actually available on this system. - * ONNX Runtime's native layer crashes (uncatchable) if we attempt CUDA - * without the required shared libraries, so we probe first. - * - * Checks both: - * 1. That system CUDA libraries (libcublasLt) are present - * 2. That onnxruntime-node ships the CUDA execution provider binary - * - * Both conditions must be true — system CUDA libs alone are not enough - * if onnxruntime-node is a CPU-only build (versions < 1.24.0). - */ -function isCudaAvailable(): boolean { - // First, verify onnxruntime-node has the CUDA provider binary. - // Without this, requesting CUDA causes an uncatchable native crash. - if (!hasOrtCudaProvider()) return false; - - // Primary: query the dynamic linker cache — covers all architectures, - // distro layouts, and custom install paths registered with ldconfig - try { - const out = execFileSync('ldconfig', ['-p'], { - timeout: 3000, - encoding: 'utf-8', - windowsHide: true, - }); - if (out.includes('libcublasLt.so.12')) return true; - } catch { - // ldconfig not available (e.g. non-standard container) - } - - // Fallback: check CUDA_PATH and LD_LIBRARY_PATH for environments where - // ldconfig doesn't know about the CUDA install (conda, manual /opt/cuda, etc.) - for (const envVar of ['CUDA_PATH', 'LD_LIBRARY_PATH']) { - const val = process.env[envVar]; - if (!val) continue; - for (const dir of val.split(':').filter(Boolean)) { - if ( - existsSync(join(dir, 'lib64', 'libcublasLt.so.12')) || - existsSync(join(dir, 'lib', 'libcublasLt.so.12')) || - existsSync(join(dir, 'libcublasLt.so.12')) - ) - return true; - } - } - - return false; -} - // Module-level state for singleton pattern let embedderInstance: FeatureExtractionPipeline | null = null; let isInitializing = false; @@ -172,7 +96,7 @@ export const initEmbedder = async ( // provider libraries are missing. DirectML stays opt-in for the same reason. // Probe for CUDA first — ONNX Runtime crashes (uncatchable native error) // if we attempt CUDA without the required shared libraries - const gpuDevice = isCudaAvailable() ? 'cuda' : 'cpu'; + const gpuDevice = isEffectiveCudaAvailable() ? 'cuda' : 'cpu'; const requestedDevice = forceDevice || (finalConfig.device === 'auto' ? gpuDevice : finalConfig.device); @@ -183,6 +107,12 @@ export const initEmbedder = async ( // Under pnpm-strict / `pnpm dlx`, transformers' phantom `onnxruntime-common` // import is unresolvable; register the fallback resolver first (#307). ensureOnnxRuntimeCommonResolvable(); + // Registered AFTER the common fallback so this hook resolves FIRST (Node + // runs the most-recently-registered hook first): on CUDA-13 hosts it + // redirects onnxruntime-node (and its version-matched onnxruntime-common) + // to the CUDA-13 build before transformers imports them. No-op on matching + // layouts, non-CUDA, Windows/DirectML, and macOS. + ensureOnnxRuntimeNodeMatchesSystem(); const { pipeline, env } = await import('@huggingface/transformers'); // Configure transformers.js environment diff --git a/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts b/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts index fbb4f4082..84b6c9fb2 100644 --- a/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts +++ b/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts @@ -21,14 +21,18 @@ * Install a synchronous, in-thread ESM resolution hook (`module.registerHooks`, * Node >= 22.15) that redirects `onnxruntime-common` to a copy gitnexus can * resolve — but only when the default resolver fails. The redirect target is - * preferentially the `onnxruntime-common` that `onnxruntime-node` (the native - * binding transformers actually loads) itself depends on, so the redirected copy - * is version-matched to that binding even under `pnpm dlx` — where gitnexus' - * npm-style `overrides` block does NOT apply, because it is honoured only from a - * root manifest and gitnexus is a transitive dependency there. It falls back to - * gitnexus' own direct `onnxruntime-common` dependency when that chain can't be - * walked. onnxruntime-common is a stable, pure-JS package whose `Tensor` surface - * is unchanged across 1.24–1.26, so either target is API-compatible. On working + * preferentially the `onnxruntime-common` that `onnxruntime-node` depends on — + * specifically {@link getEffectiveOnnxRuntimeNodeDir}, the SAME onnxruntime-node + * copy the sibling {@link ./onnxruntime-node-resolver.ts} CUDA-major redirect + * will actually load (transformers' own default when no redirect is active, + * or the CUDA-build-matched copy when one is) — so this hook and that one can + * never disagree about which onnxruntime-node's own onnxruntime-common + * dependency to pair with, even under `pnpm dlx` where gitnexus' npm-style + * `overrides` block does NOT apply (honoured only from a root manifest, and + * gitnexus is a transitive dependency there). Falls back to gitnexus' own + * direct `onnxruntime-common` dependency when that chain can't be walked. + * onnxruntime-common is a stable, pure-JS package whose `Tensor` surface is + * unchanged across 1.24–1.26, so either target is API-compatible. On working * layouts the default resolver succeeds first and the hook never fires, so * behaviour is unchanged. * @@ -55,6 +59,8 @@ */ import { registerHooks, createRequire } from 'node:module'; import { pathToFileURL } from 'node:url'; +import { join } from 'node:path'; +import { getEffectiveOnnxRuntimeNodeDir } from './onnxruntime-node-resolver.js'; import { logger } from '../logger.js'; let attempted = false; @@ -62,21 +68,19 @@ let attempted = false; /** * Compute the file: URL the hook redirects `onnxruntime-common` to. * - * Prefer the copy `onnxruntime-node` (the native binding transformers loads) - * depends on, so the redirected module is version-matched to the binding even - * under `pnpm dlx`, where transformers keeps its own pinned onnxruntime-node. - * The walk resolves transformers' MAIN entry — NOT `@huggingface/transformers/ - * package.json`, which transformers' `exports` map blocks - * (`ERR_PACKAGE_PATH_NOT_EXPORTED`) — then onnxruntime-node, then its - * onnxruntime-common. Falls back to gitnexus' own direct dependency (always - * resolvable from our scope) when any step fails. + * Pair with {@link getEffectiveOnnxRuntimeNodeDir}'s onnxruntime-node copy — + * NOT independently re-derived — so the redirected module is version-matched + * to whichever onnxruntime-node will actually load, even under `pnpm dlx` + * (where transformers keeps its own pinned onnxruntime-node) and even when + * the sibling CUDA-major redirect is active. Falls back to gitnexus' own + * direct dependency (always resolvable from our scope) when that fails. */ const resolveOnnxRuntimeCommonUrl = (): string => { const require = createRequire(import.meta.url); try { - const transformersMain = require.resolve('@huggingface/transformers'); - const ortNodePkg = createRequire(transformersMain).resolve('onnxruntime-node/package.json'); - const common = createRequire(ortNodePkg).resolve('onnxruntime-common'); + const effectiveDir = getEffectiveOnnxRuntimeNodeDir(); + if (!effectiveDir) throw new Error('no effective onnxruntime-node dir resolved'); + const common = createRequire(join(effectiveDir, 'package.json')).resolve('onnxruntime-common'); return pathToFileURL(common).href; } catch { return pathToFileURL(require.resolve('onnxruntime-common')).href; diff --git a/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts b/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts new file mode 100644 index 000000000..e9743fd8c --- /dev/null +++ b/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts @@ -0,0 +1,324 @@ +/** + * Redirect `@huggingface/transformers`' `onnxruntime-node` import to whichever + * bundled copy's CUDA build matches this host's CUDA runtime (CUDA 12 vs 13). + * + * ## Why + * transformers exact-pins `onnxruntime-node` (e.g. `1.24.3`, a CUDA **12** + * build), while gitnexus' own `onnxruntime-node: ^1.24.0` floats to the latest + * 1.x (a CUDA **13** build). npm/pnpm cannot dedupe an exact pin against a + * range, so a `npm i -g` install ends up with TWO copies: gitnexus' top-level + * CUDA-13 build (unused) and transformers' nested CUDA-12 build (the one that + * actually loads). gitnexus' `overrides` block that would collapse them is + * honoured only from a *root* manifest, so it is inert once gitnexus is a + * dependency — the same transitive-override limitation documented in + * {@link ./onnxruntime-common-resolver.ts} (#307). + * + * The consequence on a CUDA-13-only host: the nested CUDA-12 provider cannot + * find `libcublasLt.so.12`, the CUDA execution provider fails to load, and + * embeddings silently fall back to CPU (~5-6x slower) even with + * `--embedding-device cuda`. + * + * ## What this does + * Best-effort, before transformers is imported: if the system's cuBLASLt major + * (12 or 13) does NOT match the CUDA build transformers would load by default, + * but gitnexus' own top-level `onnxruntime-node` copy DOES match, install a + * synchronous ESM resolution hook (`module.registerHooks`, Node >= 22.15) that + * redirects both `onnxruntime-node` and `onnxruntime-common` to that matching + * copy. onnxruntime-common is redirected alongside so the `Tensor` surface + * stays a single identity, version-matched to the redirected binding. + * + * ## Safety + * Detection-based and conservative — it acts ONLY when it is a net improvement: + * - system CUDA major == default build major -> NO-OP (already correct) + * - no system CUDA libs / non-linux -> NO-OP (CPU path) + * - only one copy present -> NO-OP + * - neither copy matches the system -> NO-OP (never makes it worse) + * So CUDA-12 hosts, Windows (DirectML), macOS, and CPU-only hosts are + * untouched. Idempotent; any failure is swallowed and leaves the default + * resolution exactly as before. `module.registerHooks` requires Node >= 22.15 + * (the gitnexus engines floor is >= 22.0.0); on older runtimes the redirect is + * a no-op, but the default copy's CUDA major is still probed so an + * already-matching host (e.g. CUDA 12 + transformers' CUDA-12 build) keeps + * auto-selecting the GPU. + * `npm link` / symlinked local-dev checkouts are a known caveat: `resolveOurOrtNodeDir`/ + * `resolveDefaultOrtNodeDir` are anchored to this module's own real (post-symlink) + * location via `import.meta.url`, so a linked dev checkout may resolve against + * its own `node_modules` rather than the consuming app's — narrow, dev-only + * blast radius; regular npm/pnpm installs are unaffected. + * + * The CUDA-major decision is exposed via {@link getEffectiveOnnxRuntimeNodeDir} + * so the embedder's CUDA probe can inspect the SAME copy that will actually be + * loaded (the probe uses CJS `require.resolve`, which an ESM hook does not + * affect) — keeping probe and runtime consistent. + */ +import { registerHooks, createRequire } from 'node:module'; +import { pathToFileURL } from 'node:url'; +import { existsSync } from 'node:fs'; +import { join, dirname } from 'node:path'; +import { execFileSync } from 'node:child_process'; +import { logger } from '../logger.js'; + +export type CudaMajor = 12 | 13; + +const require = createRequire(import.meta.url); + +/** + * Read a shared object's NEEDED entries, tolerating ldd's non-zero exit when a + * lib is unresolved (that case still yields a usable "=> not found" stdout). + * `failed: true` means ldd produced no usable output at all (missing `ldd` + * binary, permission-denied `.so`, sandboxed exec) — distinct from "ldd ran + * fine and simply found no matching NEEDED entry" (`failed: false`, `needed: ''`), + * so callers don't have to treat "detection failed" identically to "definitely + * no CUDA provider". + */ +const readSoNeeded = (soPath: string): { needed: string; failed: boolean } => { + try { + return { + needed: execFileSync('ldd', [soPath], { + timeout: 5000, + encoding: 'utf-8', + windowsHide: true, + }), + failed: false, + }; + } catch (err) { + const out = (err as { stdout?: string } | null | undefined)?.stdout; + if (typeof out === 'string' && out.length > 0) return { needed: out, failed: false }; + return { needed: '', failed: true }; + } +}; + +/** The CUDA major an onnxruntime-node copy's CUDA provider links against, or null (Linux/x64 only ships one). */ +export const ortCudaMajor = (ortNodeDir: string): CudaMajor | null => { + const so = join( + ortNodeDir, + 'bin', + 'napi-v6', + 'linux', + process.arch, + 'libonnxruntime_providers_cuda.so', + ); + // A pre-PR CUDA-12 host relied only on this existence check (no `ldd` + // dependency) — retained here as the first, unconditional signal so a host + // whose CUDA provider `.so` is genuinely present but merely un-inspectable + // (see the `failed` case below) is never treated identically to a host that + // never shipped a CUDA provider at all. + if (!existsSync(so)) return null; + const { needed, failed } = readSoNeeded(so); + if (failed) { + logger.warn( + { so }, + 'Could not read CUDA provider dependencies (ldd failed to run) — CUDA-major detection ' + + 'is unknown, not necessarily absent; embeddings will fall back to CPU either way', + ); + } + if (/libcublasLt\.so\.13/.test(needed)) return 13; + if (/libcublasLt\.so\.12/.test(needed)) return 12; + return null; +}; + +/** The cuBLASLt major installed on this system, or null. Linux only. */ +export const detectSystemCudaMajor = (): CudaMajor | null => { + if (process.platform !== 'linux') return null; + try { + const out = execFileSync('ldconfig', ['-p'], { + timeout: 3000, + encoding: 'utf-8', + windowsHide: true, + }); + if (out.includes('libcublasLt.so.13')) return 13; + if (out.includes('libcublasLt.so.12')) return 12; + } catch { + // ldconfig not available (e.g. non-standard container) — fall through to path scan. + } + // Prefer CUDA 13 across the ENTIRE search space, not just within one + // dir/sub pair — a `.so.12` found early (e.g. a stale CUDA_PATH entry from + // a prior install) must not shadow a genuine `.so.13` found later in + // LD_LIBRARY_PATH. Return immediately on a 13 (the best possible answer); + // remember a 12 and keep scanning in case a later entry still has a 13. + let found: CudaMajor | null = null; + for (const envVar of ['CUDA_PATH', 'LD_LIBRARY_PATH']) { + const val = process.env[envVar]; + if (!val) continue; + for (const dir of val.split(':').filter(Boolean)) + for (const sub of ['lib64', 'lib', '']) + for (const maj of [13, 12] as const) + if (existsSync(join(dir, sub, `libcublasLt.so.${maj}`))) { + if (maj === 13) return 13; + found = maj; + } + } + return found; +}; + +/** onnxruntime-node dir transformers loads by default (its own nested/pinned copy). */ +const resolveDefaultOrtNodeDir = (): string | null => { + try { + const transformersMain = require.resolve('@huggingface/transformers'); + return dirname(createRequire(transformersMain).resolve('onnxruntime-node/package.json')); + } catch { + return null; + } +}; + +/** gitnexus' own direct top-level onnxruntime-node dir. */ +const resolveOurOrtNodeDir = (): string | null => { + try { + return dirname(require.resolve('onnxruntime-node/package.json')); + } catch { + return null; + } +}; + +interface Decision { + redirect: boolean; + effectiveDir: string | null; // the onnxruntime-node dir that WILL be used (default, or ours) + effectiveMajor: CudaMajor | null; // effectiveDir's own CUDA major, already probed — never re-probe it + systemMajor: CudaMajor | null; +} + +let cached: Decision | null = null; + +const decide = (): Decision => { + if (cached) return cached; + const defaultDir = resolveDefaultOrtNodeDir(); + + // Node < 22.15 has no `registerHooks` API, so a redirect can never actually + // install (see ensureOnnxRuntimeNodeMatchesSystem below) — the probe must + // agree with that up front, never reporting a redirect target that won't be + // loaded. But the DEFAULT copy still loads and needs no hook, so its CUDA + // major is still probed: a CUDA-12 host on Node 22.0–22.14 whose default + // build already matches must keep auto-selecting the GPU exactly as it did + // before this redirect existed. + const canRedirect = typeof registerHooks === 'function'; + + const systemMajor = detectSystemCudaMajor(); + // `defaultDir` resolving is NOT a precondition for checking `ourDir` below — + // if transformers' own resolution fails outright (defaultMajor stays null), + // that still counts as "the default doesn't match", so a working `ourDir` + // should still be picked up as the effective target instead of leaving + // `effectiveDir` stuck at `null`. Gated behind `systemMajor != null` (as + // before) so a non-CUDA host never pays for a provider-.so probe at all. + const defaultMajor = systemMajor != null && defaultDir ? ortCudaMajor(defaultDir) : null; + let decision: Decision = { + redirect: false, + effectiveDir: defaultDir, + effectiveMajor: defaultMajor, + systemMajor, + }; + + if (canRedirect && systemMajor != null && defaultMajor !== systemMajor) { + const ourDir = resolveOurOrtNodeDir(); + if (ourDir && ourDir !== defaultDir) { + const ourMajor = ortCudaMajor(ourDir); + if (ourMajor === systemMajor) { + decision = { redirect: true, effectiveDir: ourDir, effectiveMajor: ourMajor, systemMajor }; + } + } + } + cached = decision; + return decision; +}; + +/** + * The onnxruntime-node dir that will actually back transformers at runtime once + * {@link ensureOnnxRuntimeNodeMatchesSystem} has run — i.e. the redirected copy + * when a redirect applies, otherwise transformers' default. The CUDA probe must + * inspect THIS dir (not transformers' CJS-resolved default) so probe and + * runtime agree. Returns null only when neither copy resolves. + */ +export const getEffectiveOnnxRuntimeNodeDir = (): string | null => decide().effectiveDir; + +/** + * Whether the onnxruntime-node copy that will actually load ships a CUDA + * provider matching this host's CUDA major — reads straight from the cached + * `decide()` result rather than re-probing `ortCudaMajor`/`detectSystemCudaMajor` + * a second time (both are already computed above). `systemMajor` is checked + * for non-null explicitly so two absent majors (null === null) never count + * as a match. + */ +export const isEffectiveCudaAvailable = (): boolean => { + const d = decide(); + return d.systemMajor !== null && d.systemMajor === d.effectiveMajor; +}; + +/** + * CUDA-build-redirect status for the `doctor` Embeddings section — pure + * summary of decide()'s already-computed decision, matching + * doctor.ts's `localEmbeddingDoctorStatus`'s `{status, detail}` shape so an + * operator can tell "why is my CUDA-13 host still on CPU" apart from + * "there's no system CUDA to redirect for" at a glance. + */ +export const cudaRedirectDoctorStatus = (): { status: string; detail: string | null } => { + const d = decide(); + if (d.systemMajor === null) { + return { status: 'n/a (no system CUDA detected)', detail: null }; + } + if (d.redirect) { + return { + status: `✓ redirected onnxruntime-node to the CUDA ${d.systemMajor} build`, + detail: d.effectiveDir, + }; + } + if (d.systemMajor === d.effectiveMajor) { + return { + status: `✓ default onnxruntime-node build already matches CUDA ${d.systemMajor}`, + detail: null, + }; + } + return { + status: `✗ no CUDA ${d.systemMajor}-matched onnxruntime-node build found (falling back to CPU)`, + detail: d.effectiveDir, + }; +}; + +let attempted = false; + +/** + * Idempotently install the CUDA-build-matching redirect. Call once immediately + * before the dynamic `import('@huggingface/transformers')` on the local + * embedding path (after the runtime guard, alongside the onnxruntime-common + * fallback). No-op unless a strictly-better matching copy exists. + */ +export const ensureOnnxRuntimeNodeMatchesSystem = (): void => { + if (attempted) return; + attempted = true; + try { + if (typeof registerHooks !== 'function') return; // Node < 22.15: graceful no-op + const d = decide(); + if (!d.redirect || !d.effectiveDir) return; + + const nodeUrl = pathToFileURL( + createRequire(join(d.effectiveDir, 'package.json')).resolve('onnxruntime-node'), + ).href; + let commonUrl: string | null = null; + try { + commonUrl = pathToFileURL( + createRequire(join(d.effectiveDir, 'package.json')).resolve('onnxruntime-common'), + ).href; + } catch { + commonUrl = null; // fall back to the onnxruntime-common-resolver for common + } + + registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier === 'onnxruntime-node') return { url: nodeUrl, shortCircuit: true }; + if (commonUrl && specifier === 'onnxruntime-common') + return { url: commonUrl, shortCircuit: true }; + return nextResolve(specifier, context); + }, + }); + // info (not debug): this is the one signal an operator has that CUDA + // embeddings are actually using the GPU on this host — the common/no-op + // paths below stay at debug since they're the expected default. + logger.info( + { systemMajor: d.systemMajor, effectiveDir: d.effectiveDir }, + 'Redirected onnxruntime-node to system-matched CUDA build', + ); + } catch (err) { + logger.debug( + { err: err instanceof Error ? err.message : String(err) }, + 'onnxruntime-node CUDA-build redirect not installed', + ); + } +}; diff --git a/gitnexus/src/mcp/core/embedder.ts b/gitnexus/src/mcp/core/embedder.ts index 4dd73f317..f856446bb 100644 --- a/gitnexus/src/mcp/core/embedder.ts +++ b/gitnexus/src/mcp/core/embedder.ts @@ -23,6 +23,7 @@ import { } from '../../core/embeddings/hf-env.js'; import { getLocalEmbeddingRuntimeBlocker } from '../../core/embeddings/runtime-support.js'; import { ensureOnnxRuntimeCommonResolvable } from '../../core/embeddings/onnxruntime-common-resolver.js'; +import { ensureOnnxRuntimeNodeMatchesSystem } from '../../core/embeddings/onnxruntime-node-resolver.js'; import { silenceStdout, restoreStdout, realStderrWrite } from '../../core/lbug/pool-adapter.js'; import { logger } from '../../core/logger.js'; @@ -69,6 +70,13 @@ export const initEmbedder = async (): Promise => { // Under pnpm-strict / `pnpm dlx`, transformers' phantom `onnxruntime-common` // import is unresolvable; register the fallback resolver first (#307). ensureOnnxRuntimeCommonResolvable(); + // Registered AFTER the common fallback so this hook resolves FIRST (Node + // runs the most-recently-registered hook first): on CUDA-13 hosts it + // redirects onnxruntime-node to the system-matched build before + // transformers imports it. No-op on matching layouts, non-CUDA, + // Windows/DirectML, and macOS. Mirrors the core embedder's call site so + // MCP query-time embedding gets the same CUDA-13 fix. + ensureOnnxRuntimeNodeMatchesSystem(); const { pipeline, env } = await import('@huggingface/transformers'); env.allowLocalModels = false; diff --git a/gitnexus/test/unit/embedding-runtime-support.test.ts b/gitnexus/test/unit/embedding-runtime-support.test.ts index 5203510ec..f90c7afa8 100644 --- a/gitnexus/test/unit/embedding-runtime-support.test.ts +++ b/gitnexus/test/unit/embedding-runtime-support.test.ts @@ -21,6 +21,20 @@ vi.mock('@huggingface/transformers', () => { }; }); +/** + * Spy for the CUDA-13 build-matching resolver hook. Both local embedders must + * call this before importing transformers.js — mocked (rather than exercising + * the real resolver's env/subprocess probing) to keep this suite fast and + * platform-independent; `onnxruntime-node-resolver.test.ts` covers the + * resolver's own decision logic. + */ +const { resolverHookInstalled } = vi.hoisted(() => ({ resolverHookInstalled: vi.fn() })); + +vi.mock('../../src/core/embeddings/onnxruntime-node-resolver.js', () => ({ + ensureOnnxRuntimeNodeMatchesSystem: () => resolverHookInstalled(), + isEffectiveCudaAvailable: () => false, +})); + const EMBED_ENV_KEYS = [ 'GITNEXUS_EMBEDDING_URL', 'GITNEXUS_EMBEDDING_MODEL', @@ -44,6 +58,7 @@ const stubPlatform = (platform: NodeJS.Platform, arch: NodeJS.Architecture): (() beforeEach(() => { vi.resetModules(); transformersImported.mockClear(); + resolverHookInstalled.mockClear(); for (const key of EMBED_ENV_KEYS) delete process.env[key]; }); @@ -270,3 +285,36 @@ describe('MCP embedQuery on darwin/x64', () => { } }); }); + +describe('CUDA-13 resolver hook installation (both local-embedding entrypoints)', () => { + // Regression guard for the two local embedders drifting apart (gitnexus PR #2341 + // follow-up): both `core/embeddings/embedder.ts` and `mcp/core/embedder.ts` must + // install the CUDA-build-matching redirect during a successful local init. (The + // source itself places the call before `await import('@huggingface/transformers')` + // — not re-asserted here via mock call-order, since the hoisted `@huggingface/ + // transformers` mock's factory only fires once per file run for this external + // package, making a second per-test "called fresh" assertion on it unreliable.) + it('core embedder installs the resolver hook on a successful local init', async () => { + const restore = stubPlatform('linux', 'x64'); + try { + const { initEmbedder } = await import('../../src/core/embeddings/embedder.js'); + await expect(initEmbedder()).resolves.toBeDefined(); + + expect(resolverHookInstalled).toHaveBeenCalled(); + } finally { + restore(); + } + }); + + it('MCP embedder installs the resolver hook on a successful local init', async () => { + const restore = stubPlatform('linux', 'x64'); + try { + const { initEmbedder } = await import('../../src/mcp/core/embedder.js'); + await expect(initEmbedder()).resolves.toBeDefined(); + + expect(resolverHookInstalled).toHaveBeenCalled(); + } finally { + restore(); + } + }); +}); diff --git a/gitnexus/test/unit/hooks.test.ts b/gitnexus/test/unit/hooks.test.ts index 91f18232c..b3bc3db97 100644 --- a/gitnexus/test/unit/hooks.test.ts +++ b/gitnexus/test/unit/hooks.test.ts @@ -356,8 +356,16 @@ describe('windowsHide regression', () => { ['gitnexus/src/cli/setup.ts', path.resolve(__dirname, '..', '..', 'src', 'cli', 'setup.ts')], ['gitnexus/src/cli/wiki.ts', path.resolve(__dirname, '..', '..', 'src', 'cli', 'wiki.ts')], [ - 'gitnexus/src/core/embeddings/embedder.ts', - path.resolve(__dirname, '..', '..', 'src', 'core', 'embeddings', 'embedder.ts'), + 'gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts', + path.resolve( + __dirname, + '..', + '..', + 'src', + 'core', + 'embeddings', + 'onnxruntime-node-resolver.ts', + ), ], [ 'gitnexus/src/core/git-staleness.ts', diff --git a/gitnexus/test/unit/onnxruntime-common-resolver.test.ts b/gitnexus/test/unit/onnxruntime-common-resolver.test.ts index 801cb001e..00cc86a46 100644 --- a/gitnexus/test/unit/onnxruntime-common-resolver.test.ts +++ b/gitnexus/test/unit/onnxruntime-common-resolver.test.ts @@ -17,13 +17,27 @@ const RESOLVER = '../../src/core/embeddings/onnxruntime-common-resolver.js'; * (Re)load the resolver with a chosen `registerHooks` mocked into node:module. * `vi.resetModules()` + the fresh `import()` re-initialises the module-level * one-shot guard, so each test gets a pristine resolver with no shared state. + * + * When `getEffectiveOnnxRuntimeNodeDir` is supplied, the sibling + * onnxruntime-node-resolver.js is also mocked with it — letting a test drive + * (or spy on) whichever onnxruntime-node dir this hook's own onnxruntime-common + * lookup defers to, instead of independently re-deriving transformers' + * default (#2341 follow-up). */ -async function loadResolver(registerHooks: unknown) { +async function loadResolver( + registerHooks: unknown, + getEffectiveOnnxRuntimeNodeDir?: () => string | null, +) { vi.resetModules(); vi.doMock('node:module', async (importOriginal) => { const orig = await importOriginal(); return { ...orig, registerHooks }; }); + if (getEffectiveOnnxRuntimeNodeDir) { + vi.doMock('../../src/core/embeddings/onnxruntime-node-resolver.js', () => ({ + getEffectiveOnnxRuntimeNodeDir, + })); + } return import(RESOLVER); } @@ -36,6 +50,7 @@ const moduleNotFound = (): Error => { afterEach(() => { vi.doUnmock('node:module'); + vi.doUnmock('../../src/core/embeddings/onnxruntime-node-resolver.js'); }); describe('ensureOnnxRuntimeCommonResolvable — installation', () => { @@ -70,9 +85,9 @@ describe('ensureOnnxRuntimeCommonResolvable — installation', () => { describe('ensureOnnxRuntimeCommonResolvable — resolve hook behaviour', () => { /** Install the fallback and return the resolve closure handed to registerHooks. */ - async function captureResolve() { + async function captureResolve(getEffectiveOnnxRuntimeNodeDir?: () => string | null) { const spy = vi.fn(); - const mod = await loadResolver(spy); + const mod = await loadResolver(spy, getEffectiveOnnxRuntimeNodeDir); mod.ensureOnnxRuntimeCommonResolvable(); return spy.mock.calls[0][0].resolve as ( s: string, @@ -129,4 +144,23 @@ describe('ensureOnnxRuntimeCommonResolvable — resolve hook behaviour', () => { expect(() => resolve('onnxruntime-common', ctx, next)).toThrow(err); }); + + it("defers to getEffectiveOnnxRuntimeNodeDir() instead of independently re-deriving transformers' default (#2341 follow-up)", async () => { + const effectiveDirSpy = vi.fn(() => null as string | null); + const resolve = await captureResolve(effectiveDirSpy); + const next = vi.fn(() => { + throw moduleNotFound(); + }); + + const res = resolve('onnxruntime-common', ctx, next) as { url: string; shortCircuit: boolean }; + + // The sibling module's decision is consulted (not bypassed)... + expect(effectiveDirSpy).toHaveBeenCalled(); + // ...and since it reported no effective dir here, the code falls back to + // gitnexus' own direct dependency (the same fallback as "default + // resolution fails") rather than independently re-deriving a different + // path from @huggingface/transformers on its own. + expect(res.shortCircuit).toBe(true); + expect(res.url).toMatch(/^file:\/\/.*\/node_modules\/onnxruntime-common\/.*\.js$/); + }); }); diff --git a/gitnexus/test/unit/onnxruntime-node-resolver.test.ts b/gitnexus/test/unit/onnxruntime-node-resolver.test.ts new file mode 100644 index 000000000..91711a98a --- /dev/null +++ b/gitnexus/test/unit/onnxruntime-node-resolver.test.ts @@ -0,0 +1,759 @@ +import { describe, it, expect, vi, afterEach } from 'vitest'; +import path from 'node:path'; + +/** + * Tests for the CUDA-build-matching onnxruntime-node redirect. + * + * `@huggingface/transformers` exact-pins a CUDA-12 `onnxruntime-node`, while + * gitnexus' own dep floats to a CUDA-13 build; on a CUDA-13 host this module + * redirects transformers to the matching copy so embeddings use the GPU instead + * of silently falling back to CPU. The detection primitives (`ldconfig` / `ldd` + * / path scan) and `module.registerHooks` are mocked so the pure decision logic + * is asserted without touching the real loader or the host's CUDA install. + */ + +const RESOLVER = '../../src/core/embeddings/onnxruntime-node-resolver.js'; + +const REAL_PLATFORM = process.platform; +const REAL_ENV = { ...process.env }; + +/** + * Node's `path` module is bound to `path.win32` or `path.posix` based on the + * REAL host OS at process start — stubbing `process.platform` later (as this + * file's tests do, for the resolver's OWN platform branching) has no effect + * on it. So on a genuine Windows CI runner, the resolver's `join(...)` calls + * normalize our forward-slash fake dirs to backslash-separated strings, + * which would silently fail to match the forward-slash fixtures/prefixes + * below. Normalize before every comparison so these tests are host-OS-agnostic. + */ +const toPosix = (p: string): string => p.replace(/\\/g, '/'); + +/** Three fake, distinct onnxruntime-node locations for driving decide() into redirect:true. */ +interface FakeDirs { + /** gitnexus' own top-level onnxruntime-node dir (resolved via the module's own require). */ + ourDir: string; + /** transformers' pinned/nested onnxruntime-node dir (resolved via createRequire(transformersMain)). */ + defaultDir: string; + /** fake resolved path for require.resolve('@huggingface/transformers'). */ + transformersMain: string; + /** When false, createRequire(transformersMain).resolve('onnxruntime-node/package.json') throws + * (simulating resolveDefaultOrtNodeDir() failing outright) instead of resolving to `defaultDir`. */ + defaultResolvable?: boolean; +} + +interface LoadOpts { + registerHooks?: unknown; + platform?: NodeJS.Platform; + execFileSync?: (cmd: string, args: string[]) => string; + existsSync?: (p: string) => boolean; + fakeDirs?: FakeDirs; + /** Force the resolver's `join`/`dirname` calls to use `path.win32` semantics + * (backslash-normalized output) regardless of the real host OS — proves the + * `toPosix()` normalization above actually works, rather than merely being + * argued for (#2341 follow-up). */ + forceWin32Path?: boolean; +} + +/** A require()-like function whose .resolve() is driven entirely by a specifier -> path map. */ +function fakeRequire(resolveMap: Record) { + return Object.assign( + (specifier: string) => { + throw new Error(`fakeRequire: unexpected require(${specifier})`); + }, + { + resolve: (specifier: string) => { + const hit = resolveMap[specifier]; + if (!hit) { + throw Object.assign(new Error(`Cannot find module '${specifier}'`), { + code: 'MODULE_NOT_FOUND', + }); + } + return hit; + }, + }, + ); +} + +/** + * (Re)load the resolver with detection primitives + `registerHooks` mocked. + * `vi.resetModules()` clears the module-level decision cache and one-shot guard, + * so each test gets a pristine resolver. + * + * When `fakeDirs` is supplied, `createRequire` is also mocked so the module's + * two CJS resolve-walks (`resolveOurOrtNodeDir`/`resolveDefaultOrtNodeDir`, and + * the nodeUrl/commonUrl lookup inside `ensureOnnxRuntimeNodeMatchesSystem`) each + * resolve against a distinct fake directory instead of whatever's actually + * installed in this test's real node_modules — the only way to drive + * `decide() -> redirect:true` deterministically without touching production code. + */ +async function loadResolver(opts: LoadOpts = {}) { + vi.resetModules(); + // Destructuring defaults (`= vi.fn()`) only apply when the property is + // `undefined` — but callers pass `registerHooks: undefined` specifically to + // simulate Node < 22.15 (no synchronous-hooks API), so a plain destructuring + // default would silently substitute a real mock function and defeat that. + // `'registerHooks' in opts` distinguishes "omitted → default to a spy" from + // "explicitly undefined → simulate its absence". + const registerHooks = 'registerHooks' in opts ? opts.registerHooks : vi.fn(); + const { + platform = 'linux', + execFileSync = () => { + throw Object.assign(new Error('enoent'), { code: 'ENOENT' }); + }, + existsSync = () => false, + fakeDirs, + forceWin32Path = false, + } = opts; + + if (forceWin32Path) { + vi.doMock('node:path', () => ({ ...path.win32, default: path.win32 })); + } + + vi.doMock('node:module', async (io) => { + const orig = await io(); + if (!fakeDirs) return { ...orig, registerHooks }; + + const ourRequire = fakeRequire({ + '@huggingface/transformers': fakeDirs.transformersMain, + 'onnxruntime-node/package.json': `${fakeDirs.ourDir}/package.json`, + }); + const defaultRequire = fakeRequire( + fakeDirs.defaultResolvable === false + ? {} + : { 'onnxruntime-node/package.json': `${fakeDirs.defaultDir}/package.json` }, + ); + const effectiveRequire = fakeRequire({ + 'onnxruntime-node': `${fakeDirs.ourDir}/index.js`, + 'onnxruntime-common': `${fakeDirs.ourDir}/node_modules/onnxruntime-common/index.js`, + }); + return { + ...orig, + registerHooks, + createRequire: (from: string) => { + // `from` is produced by the resolver's own `join(effectiveDir, 'package.json')` + // call — backslash-normalized on a real Windows host even though + // `fakeDirs.ourDir` etc. are forward-slash fixtures; normalize before comparing. + const normalizedFrom = toPosix(from); + if (normalizedFrom === fakeDirs.transformersMain) return defaultRequire; + if (normalizedFrom === `${fakeDirs.ourDir}/package.json`) return effectiveRequire; + return ourRequire; + }, + }; + }); + vi.doMock('node:child_process', async (io) => ({ + ...(await io()), + // Normalize args (the `.so` path for `ldd`) so callers' forward-slash + // prefix checks match regardless of which path module the resolver's + // own `join(...)` calls were bound to on the host running this test. + execFileSync: (cmd: string, args: string[]) => execFileSync(cmd, args.map(toPosix)), + })); + vi.doMock('node:fs', async (io) => ({ + ...(await io()), + existsSync: (p: unknown) => existsSync(toPosix(String(p))), + })); + + Object.defineProperty(process, 'platform', { value: platform, configurable: true }); + return import(RESOLVER); +} + +afterEach(() => { + vi.doUnmock('node:module'); + vi.doUnmock('node:child_process'); + vi.doUnmock('node:fs'); + vi.doUnmock('node:path'); + Object.defineProperty(process, 'platform', { value: REAL_PLATFORM, configurable: true }); + process.env = { ...REAL_ENV }; +}); + +describe('detectSystemCudaMajor', () => { + it.each(['darwin', 'win32'] as const)( + 'returns null on non-linux platforms (%s)', + async (platform) => { + const mod = await loadResolver({ platform }); + expect(mod.detectSystemCudaMajor()).toBeNull(); + }, + ); + + it('prefers CUDA 13 over 12 when ldconfig lists both', async () => { + const mod = await loadResolver({ + execFileSync: () => + 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13\n' + + 'libcublasLt.so.12 (libc6,x86-64) => /old/libcublasLt.so.12', + }); + expect(mod.detectSystemCudaMajor()).toBe(13); + }); + + it('detects CUDA 12 when only .so.12 is present', async () => { + const mod = await loadResolver({ + execFileSync: () => 'libcublasLt.so.12 (libc6,x86-64) => /usr/lib/libcublasLt.so.12', + }); + expect(mod.detectSystemCudaMajor()).toBe(12); + }); + + it('falls back to an LD_LIBRARY_PATH scan when ldconfig is unavailable', async () => { + process.env.LD_LIBRARY_PATH = '/opt/cuda/lib64'; + const mod = await loadResolver({ + execFileSync: () => { + throw new Error('ldconfig missing'); + }, + existsSync: (p) => p === '/opt/cuda/lib64/libcublasLt.so.13', + }); + expect(mod.detectSystemCudaMajor()).toBe(13); + }); + + it('returns null when no cuBLASLt is found anywhere', async () => { + const mod = await loadResolver({ execFileSync: () => 'libfoo.so => /x/libfoo.so' }); + expect(mod.detectSystemCudaMajor()).toBeNull(); + }); + + it('falls back to a CUDA_PATH scan when ldconfig is unavailable (#2341 follow-up)', async () => { + // Mirrors the existing LD_LIBRARY_PATH-only test above — CUDA_PATH is + // scanned first in the fallback loop and was previously untested on its own. + process.env.CUDA_PATH = '/opt/cuda'; + const mod = await loadResolver({ + execFileSync: () => { + throw new Error('ldconfig missing'); + }, + existsSync: (p) => p === '/opt/cuda/lib64/libcublasLt.so.13', + }); + expect(mod.detectSystemCudaMajor()).toBe(13); + }); + + it('returns null (not a false match) when the ldconfig output is garbled/unrecognized', async () => { + const mod = await loadResolver({ + execFileSync: () => 'some-corrupted-binary-output-\x00\xff-not-a-cuda-lib-line', + }); + expect(mod.detectSystemCudaMajor()).toBeNull(); + }); + + it('prefers a CUDA 13 found later in the search path over a CUDA 12 found earlier (#2341 follow-up)', async () => { + // A stale CUDA_PATH entry (e.g. left over from a prior install) only has + // .so.12; LD_LIBRARY_PATH, scanned after it, has the genuine .so.13. The + // scan must not stop at the first match — it must keep looking for a + // better (13) answer across the WHOLE search space. + process.env.CUDA_PATH = '/opt/old-cuda-12'; + process.env.LD_LIBRARY_PATH = '/opt/cuda-13/lib64'; + const mod = await loadResolver({ + execFileSync: () => { + throw new Error('ldconfig missing'); + }, + existsSync: (p) => + p === '/opt/old-cuda-12/libcublasLt.so.12' || p === '/opt/cuda-13/lib64/libcublasLt.so.13', + }); + expect(mod.detectSystemCudaMajor()).toBe(13); + }); +}); + +describe('ortCudaMajor', () => { + it('returns null when the CUDA provider .so is absent', async () => { + const mod = await loadResolver({ existsSync: () => false }); + expect(mod.ortCudaMajor('/pkg/onnxruntime-node')).toBeNull(); + }); + + it('reads CUDA 13 from the provider .so NEEDED entries', async () => { + const mod = await loadResolver({ + existsSync: () => true, + execFileSync: () => 'libcublasLt.so.13 => /usr/local/cuda/lib64/libcublasLt.so.13', + }); + expect(mod.ortCudaMajor('/pkg/onnxruntime-node')).toBe(13); + }); + + it('reads CUDA 12 even when the NEEDED lib is unresolved (ldd non-zero exit)', async () => { + const mod = await loadResolver({ + existsSync: () => true, + execFileSync: () => { + // ldd exits non-zero with the "=> not found" line on stdout + throw Object.assign(new Error('ldd failed'), { + stdout: 'libcublasLt.so.12 => not found', + }); + }, + }); + expect(mod.ortCudaMajor('/pkg/onnxruntime-node')).toBe(12); + }); + + it('returns null (not a false match) when the ldd output is garbled/unrecognized (#2341 follow-up)', async () => { + const mod = await loadResolver({ + existsSync: () => true, + execFileSync: () => 'libunrelated.so.1 => /x/libunrelated.so.1\nlibc.so.6 => /lib/libc.so.6', + }); + expect(mod.ortCudaMajor('/pkg/onnxruntime-node')).toBeNull(); + }); + + it('warns (detection failed) when ldd produces no usable output at all, distinct from the silent no-provider case (#2341 follow-up)', async () => { + // Capture AFTER loadResolver() so the capture targets the same (freshly + // reset) logger.js instance the resolver module itself imports — the + // module registry is cleared by loadResolver()'s vi.resetModules(). + const mod = await loadResolver({ + existsSync: () => true, + // Simulates a missing `ldd` binary (ENOENT) or a permission-denied + // `.so`: execFileSync throws with no `stdout` at all, unlike the + // "=> not found" case above which still yields usable text. + execFileSync: () => { + throw Object.assign(new Error('spawn ldd ENOENT'), { code: 'ENOENT' }); + }, + }); + const { _captureLogger } = await import('../../src/core/logger.js'); + const cap = _captureLogger(); + try { + expect(mod.ortCudaMajor('/pkg/onnxruntime-node')).toBeNull(); + + const records = cap.records(); + expect( + records.some((r) => r.msg?.includes('Could not read CUDA provider dependencies')), + ).toBe(true); + } finally { + cap.restore(); + } + }); + + it('does not warn when the CUDA provider .so is simply absent (no detection was even attempted)', async () => { + const mod = await loadResolver({ existsSync: () => false }); + const { _captureLogger } = await import('../../src/core/logger.js'); + const cap = _captureLogger(); + try { + expect(mod.ortCudaMajor('/pkg/onnxruntime-node')).toBeNull(); + + const records = cap.records(); + expect( + records.some((r) => r.msg?.includes('Could not read CUDA provider dependencies')), + ).toBe(false); + } finally { + cap.restore(); + } + }); +}); + +describe('ensureOnnxRuntimeNodeMatchesSystem', () => { + it('no-ops gracefully when registerHooks is unavailable (Node < 22.15), leaving the module otherwise functional', async () => { + const mod = await loadResolver({ registerHooks: undefined }); + expect(() => mod.ensureOnnxRuntimeNodeMatchesSystem()).not.toThrow(); + // The one-shot guard tripping (or not) must not corrupt decide()'s cache — + // subsequent calls to the other exports still work normally afterward. + expect(() => mod.getEffectiveOnnxRuntimeNodeDir()).not.toThrow(); + expect(mod.isEffectiveCudaAvailable()).toBe(false); // redirect can never be active without registerHooks + }); + + it('installs no hook when there is no system CUDA (no redirect needed)', async () => { + const spy = vi.fn(); + // non-linux → detectSystemCudaMajor() === null → decide() → redirect: false + const mod = await loadResolver({ registerHooks: spy, platform: 'darwin' }); + mod.ensureOnnxRuntimeNodeMatchesSystem(); + expect(spy).not.toHaveBeenCalled(); + }); + + it('is idempotent in the no-redirect case: a second call is still a no-op (registerHooks never called)', async () => { + const spy = vi.fn(); + const mod = await loadResolver({ registerHooks: spy, platform: 'darwin' }); + mod.ensureOnnxRuntimeNodeMatchesSystem(); + mod.ensureOnnxRuntimeNodeMatchesSystem(); + // (True install-once idempotency, where a redirect WOULD fire without the + // guard, is covered by "installs registerHooks exactly once when the + // redirect is active" below — this case only proves repeated calls stay + // side-effect-free when there's nothing to install.) + expect(spy).not.toHaveBeenCalled(); + }); + + it('exposes an effective onnxruntime-node dir (string or null) for the CUDA probe, never throwing', async () => { + const mod = await loadResolver({ platform: 'darwin' }); + // Non-linux: no redirect, so the effective dir is transformers' default — + // a string when resolvable in the test tree (it really is, in this repo), + // or null if resolution ever genuinely fails. + let result: string | null | undefined; + expect(() => { + result = mod.getEffectiveOnnxRuntimeNodeDir(); + }).not.toThrow(); + expect(result === null || typeof result === 'string').toBe(true); + }); +}); + +describe('decide() — registerHooks gating (#2341 follow-up)', () => { + // ensureOnnxRuntimeNodeMatchesSystem() can never install a redirect on + // Node < 22.15 (no registerHooks), so decide() must never report `ourDir` + // as the effective target there. But transformers' DEFAULT copy still loads + // without any hook, so its CUDA major must still be probed: a CUDA-12 host + // on Node 22.0–22.14 whose default build already matches has to keep the + // GPU it auto-selected before this redirect existed (pre-PR + // isCudaAvailable() behavior), not silently fall back to CPU. + const fakeDirs = { + ourDir: '/fake/our/onnxruntime-node', + defaultDir: '/fake/transformers-nested/onnxruntime-node', + transformersMain: '/fake/transformers/dist/transformers.node.mjs', + }; + const soPrefix = (dir: string) => `${dir}/bin/napi-v6/linux`; + + // System CUDA major is the parameter; the two bundled copies are fixed at + // ours=13 / default=12 (the PR's own documented layout). + function loadOldNodeResolver(systemMajor: 12 | 13) { + return loadResolver({ + registerHooks: undefined, + platform: 'linux', + fakeDirs, + existsSync: (p) => + p.startsWith(soPrefix(fakeDirs.ourDir)) || p.startsWith(soPrefix(fakeDirs.defaultDir)), + execFileSync: (cmd, args) => { + if (cmd === 'ldconfig') + return `libcublasLt.so.${systemMajor} (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.${systemMajor}`; + if (cmd === 'ldd') { + const target = args[0] ?? ''; + if (target.startsWith(soPrefix(fakeDirs.ourDir))) + return 'libcublasLt.so.13 => /usr/local/cuda-13/lib64/libcublasLt.so.13'; + if (target.startsWith(soPrefix(fakeDirs.defaultDir))) + return 'libcublasLt.so.12 => /usr/local/cuda-12/lib64/libcublasLt.so.12'; + } + throw new Error(`unexpected execFileSync(${cmd}, ${JSON.stringify(args)})`); + }, + }); + } + + it('never reports a redirect target that cannot be installed (CUDA-13 host, mismatched default)', async () => { + const mod = await loadOldNodeResolver(13); + expect(toPosix(String(mod.getEffectiveOnnxRuntimeNodeDir()))).toBe(fakeDirs.defaultDir); + expect(mod.isEffectiveCudaAvailable()).toBe(false); + expect(() => mod.ensureOnnxRuntimeNodeMatchesSystem()).not.toThrow(); + }); + + it('still probes the default copy: a CUDA-12 host whose default build matches keeps the GPU', async () => { + const mod = await loadOldNodeResolver(12); + expect(toPosix(String(mod.getEffectiveOnnxRuntimeNodeDir()))).toBe(fakeDirs.defaultDir); + expect(mod.isEffectiveCudaAvailable()).toBe(true); + }); +}); + +describe('ensureOnnxRuntimeNodeMatchesSystem — redirect:true (#2341 follow-up)', () => { + // The prior test suite never drove decide() into redirect:true (it never + // faked createRequire), so the actual installed resolve() closure — the PR's + // real shipped behavior — had zero test coverage. Reproduce the PR's own + // documented common case: system has CUDA 13, transformers' default build is + // CUDA 12, gitnexus' own top-level build is CUDA 13. + const fakeDirs = { + ourDir: '/fake/our/onnxruntime-node', + defaultDir: '/fake/transformers-nested/onnxruntime-node', + transformersMain: '/fake/transformers/dist/transformers.node.mjs', + }; + + const soPath = (dir: string) => `${dir}/bin/napi-v6/linux`; // arch-agnostic prefix match below + + function loadRedirectActiveResolver(registerHooksSpy: unknown) { + return loadResolver({ + registerHooks: registerHooksSpy, + platform: 'linux', + fakeDirs, + existsSync: (p) => + p.startsWith(soPath(fakeDirs.ourDir)) || p.startsWith(soPath(fakeDirs.defaultDir)), + execFileSync: (cmd, args) => { + if (cmd === 'ldconfig') + return 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13'; + if (cmd === 'ldd') { + const target = args[0] ?? ''; + if (target.startsWith(soPath(fakeDirs.ourDir))) { + return 'libcublasLt.so.13 => /usr/local/cuda-13/lib64/libcublasLt.so.13'; + } + if (target.startsWith(soPath(fakeDirs.defaultDir))) { + return 'libcublasLt.so.12 => /usr/local/cuda-12/lib64/libcublasLt.so.12'; + } + } + throw new Error(`unexpected execFileSync(${cmd}, ${JSON.stringify(args)})`); + }, + }); + } + + it('reports the redirect-active effective dir as our own CUDA-13 build', async () => { + const mod = await loadRedirectActiveResolver(vi.fn()); + expect(mod.getEffectiveOnnxRuntimeNodeDir()).toBe(fakeDirs.ourDir); + }); + + it('installs registerHooks exactly once when the redirect is active', async () => { + const spy = vi.fn(); + const mod = await loadRedirectActiveResolver(spy); + mod.ensureOnnxRuntimeNodeMatchesSystem(); + expect(spy).toHaveBeenCalledTimes(1); + expect(typeof spy.mock.calls[0][0].resolve).toBe('function'); + }); + + it('the installed resolve() closure redirects onnxruntime-node and onnxruntime-common, and passes through everything else', async () => { + const spy = vi.fn(); + const mod = await loadRedirectActiveResolver(spy); + mod.ensureOnnxRuntimeNodeMatchesSystem(); + const resolve = spy.mock.calls[0][0].resolve as ( + s: string, + c: never, + n: (s: string, c: never) => unknown, + ) => unknown; + const ctx = {} as never; + const next = vi.fn(() => ({ url: 'file:///should-not-be-used', shortCircuit: true })); + + const nodeResult = resolve('onnxruntime-node', ctx, next) as { + url: string; + shortCircuit: boolean; + }; + expect(nodeResult).toEqual({ + url: expect.stringContaining('/fake/our/onnxruntime-node/index.js'), + shortCircuit: true, + }); + expect(next).not.toHaveBeenCalled(); + + const commonResult = resolve('onnxruntime-common', ctx, next) as { + url: string; + shortCircuit: boolean; + }; + expect(commonResult).toEqual({ + url: expect.stringContaining( + '/fake/our/onnxruntime-node/node_modules/onnxruntime-common/index.js', + ), + shortCircuit: true, + }); + expect(next).not.toHaveBeenCalled(); + + resolve('some-other-package', ctx, next); + expect(next).toHaveBeenCalledWith('some-other-package', ctx); + }); + + it('isEffectiveCudaAvailable() reports true when the redirect-active effective build matches the system', async () => { + const mod = await loadRedirectActiveResolver(vi.fn()); + expect(mod.isEffectiveCudaAvailable()).toBe(true); + }); + + it('logs the successful redirect at info level (#2341 follow-up)', async () => { + const mod = await loadRedirectActiveResolver(vi.fn()); + const { _captureLogger } = await import('../../src/core/logger.js'); + const cap = _captureLogger(); + try { + mod.ensureOnnxRuntimeNodeMatchesSystem(); + const record = cap + .records() + .find((r) => r.msg?.includes('Redirected onnxruntime-node to system-matched CUDA build')); + expect(record).toBeDefined(); + expect(record?.level).toBe(30); // pino 'info' + } finally { + cap.restore(); + } + }); + + it('does not log at info when no redirect is needed (common, expected path)', async () => { + // Non-linux -> no system CUDA -> decide() never redirects. + const mod = await loadResolver({ registerHooks: vi.fn(), platform: 'darwin' }); + const { _captureLogger } = await import('../../src/core/logger.js'); + const cap = _captureLogger('debug'); // capture below the default 'info' to prove nothing else fires either + try { + mod.ensureOnnxRuntimeNodeMatchesSystem(); + const infoOrAboveRecords = cap.records().filter((r) => (r.level ?? 0) >= 30); + expect(infoOrAboveRecords).toHaveLength(0); + } finally { + cap.restore(); + } + }); +}); + +describe('cudaRedirectDoctorStatus (#2341 follow-up)', () => { + it('reports n/a when there is no system CUDA', async () => { + const mod = await loadResolver({ registerHooks: vi.fn(), platform: 'darwin' }); + expect(mod.cudaRedirectDoctorStatus()).toEqual({ + status: 'n/a (no system CUDA detected)', + detail: null, + }); + }); + + it('reports the redirect-active status with the effective dir as detail', async () => { + const fakeDirs = { + ourDir: '/fake/our/onnxruntime-node-doctor', + defaultDir: '/fake/transformers-nested/onnxruntime-node-doctor', + transformersMain: '/fake/transformers/doctor/index.js', + }; + const soPath = (dir: string) => `${dir}/bin/napi-v6/linux`; + const mod = await loadResolver({ + registerHooks: vi.fn(), + platform: 'linux', + fakeDirs, + existsSync: (p) => + p.startsWith(soPath(fakeDirs.ourDir)) || p.startsWith(soPath(fakeDirs.defaultDir)), + execFileSync: (cmd, args) => { + if (cmd === 'ldconfig') + return 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13'; + if (cmd === 'ldd') { + const target = args[0] ?? ''; + if (target.startsWith(soPath(fakeDirs.ourDir))) + return 'libcublasLt.so.13 => /a/libcublasLt.so.13'; + if (target.startsWith(soPath(fakeDirs.defaultDir))) + return 'libcublasLt.so.12 => /a/libcublasLt.so.12'; + } + throw new Error(`unexpected execFileSync(${cmd}, ${JSON.stringify(args)})`); + }, + }); + + expect(mod.cudaRedirectDoctorStatus()).toEqual({ + status: expect.stringContaining('redirected onnxruntime-node to the CUDA 13 build'), + detail: fakeDirs.ourDir, + }); + }); + + it('reports a mismatch status (with no fix available) when neither copy ships a matching CUDA provider', async () => { + const mod = await loadResolver({ + registerHooks: vi.fn(), + platform: 'linux', + existsSync: () => false, // no onnxruntime-node copy ships a CUDA provider .so at all + execFileSync: (cmd) => { + if (cmd === 'ldconfig') + return 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13'; + throw new Error('ldd should not be reached when existsSync is false'); + }, + }); + + // `detail` (the resolved effectiveDir) isn't asserted here — resolveDefaultOrtNodeDir() + // isn't mocked in this test, so it resolves against this sandbox's real + // node_modules and its exact value isn't the point of this case; the + // redirect-active test above already covers `detail` precisely. + expect(mod.cudaRedirectDoctorStatus().status).toContain( + 'no CUDA 13-matched onnxruntime-node build found', + ); + }); +}); + +describe('isEffectiveCudaAvailable — no redundant subprocess spawns (#2341 follow-up)', () => { + it('probes ldconfig/ldd only once total, regardless of how many times the effective dir and CUDA match are queried', async () => { + const fakeDirs = { + ourDir: '/fake/our/onnxruntime-node-u8', + defaultDir: '/fake/transformers-nested/onnxruntime-node-u8', + transformersMain: '/fake/transformers/u8/index.js', + }; + const soPath = (dir: string) => `${dir}/bin/napi-v6/linux`; + const existsSyncSpy = vi.fn( + (p: string) => + p.startsWith(soPath(fakeDirs.ourDir)) || p.startsWith(soPath(fakeDirs.defaultDir)), + ); + const execFileSyncSpy = vi.fn((cmd: string, args: string[]) => { + if (cmd === 'ldconfig') + return 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13'; + if (cmd === 'ldd') { + const target = args[0] ?? ''; + if (target.startsWith(soPath(fakeDirs.ourDir))) { + return 'libcublasLt.so.13 => /usr/local/cuda-13/lib64/libcublasLt.so.13'; + } + if (target.startsWith(soPath(fakeDirs.defaultDir))) { + return 'libcublasLt.so.12 => /usr/local/cuda-12/lib64/libcublasLt.so.12'; + } + } + throw new Error(`unexpected execFileSync(${cmd}, ${JSON.stringify(args)})`); + }); + + const mod = await loadResolver({ + registerHooks: vi.fn(), + platform: 'linux', + fakeDirs, + existsSync: existsSyncSpy, + execFileSync: execFileSyncSpy, + }); + + // Query the decision through both public entry points, each more than once. + mod.getEffectiveOnnxRuntimeNodeDir(); + mod.isEffectiveCudaAvailable(); + mod.getEffectiveOnnxRuntimeNodeDir(); + expect(mod.isEffectiveCudaAvailable()).toBe(true); + + // decide() is memoized: exactly one ldconfig call (system major) and one + // ldd call per onnxruntime-node dir actually probed (default + ours) — + // never re-invoked across the 4 queries above. + const ldconfigCalls = execFileSyncSpy.mock.calls.filter(([cmd]) => cmd === 'ldconfig'); + const lddCalls = execFileSyncSpy.mock.calls.filter(([cmd]) => cmd === 'ldd'); + expect(ldconfigCalls).toHaveLength(1); + expect(lddCalls).toHaveLength(2); // defaultDir once, ourDir once + }); +}); + +describe('decide() — ourDir checked independently of defaultDir (#2341 follow-up)', () => { + it("picks ourDir as the effective target when transformers' own onnxruntime-node resolution fails outright", async () => { + const fakeDirs = { + ourDir: '/fake/our/onnxruntime-node-u5', + defaultDir: '/fake/unreachable/onnxruntime-node-u5', + transformersMain: '/fake/transformers/u5/index.js', + defaultResolvable: false, // createRequire(transformersMain).resolve(...) throws -> defaultDir stays null + }; + const soPrefix = (dir: string) => `${dir}/bin/napi-v6/linux`; + + const mod = await loadResolver({ + registerHooks: vi.fn(), + platform: 'linux', + fakeDirs, + existsSync: (p) => p.startsWith(soPrefix(fakeDirs.ourDir)), + execFileSync: (cmd, args) => { + if (cmd === 'ldconfig') { + return 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13'; + } + if (cmd === 'ldd' && (args[0] ?? '').startsWith(soPrefix(fakeDirs.ourDir))) { + return 'libcublasLt.so.13 => /usr/local/cuda-13/lib64/libcublasLt.so.13'; + } + throw new Error(`unexpected execFileSync(${cmd}, ${JSON.stringify(args)})`); + }, + }); + + // Before this fix, the ourDir fallback lookup was nested inside + // `if (systemMajor != null && defaultDir)`, so a null defaultDir skipped + // checking ourDir entirely and this would incorrectly return null. + expect(mod.getEffectiveOnnxRuntimeNodeDir()).toBe(fakeDirs.ourDir); + }); +}); + +describe('cross-platform path handling (#2341 follow-up)', () => { + // This test file was added to cross-platform-tests.ts's PLATFORM_LOGIC list + // (so it now runs on the Windows CI matrix, not just Ubuntu). Node's `path` + // module is bound to path.win32 on a real Windows host regardless of any + // process.platform stub — so the resolver's own join(effectiveDir, + // 'package.json') calls backslash-normalize even when these tests fake + // platform: 'linux'. forceWin32Path proves the toPosix() normalization + // added above actually handles that, rather than merely being argued for. + const fakeDirs = { + ourDir: '/fake/our/onnxruntime-node', + defaultDir: '/fake/transformers-nested/onnxruntime-node', + transformersMain: '/fake/transformers/dist/transformers.node.mjs', + }; + const soPath = (dir: string) => `${dir}/bin/napi-v6/linux`; + + it('resolves the redirect-active dir and installs the resolve() closure correctly even when join()/dirname() backslash-normalize (simulated real Windows)', async () => { + const spy = vi.fn(); + const mod = await loadResolver({ + registerHooks: spy, + platform: 'linux', + fakeDirs, + forceWin32Path: true, + existsSync: (p) => + p.startsWith(soPath(fakeDirs.ourDir)) || p.startsWith(soPath(fakeDirs.defaultDir)), + execFileSync: (cmd, args) => { + if (cmd === 'ldconfig') + return 'libcublasLt.so.13 (libc6,x86-64) => /usr/local/cuda/lib64/libcublasLt.so.13'; + if (cmd === 'ldd') { + const target = args[0] ?? ''; + if (target.startsWith(soPath(fakeDirs.ourDir))) + return 'libcublasLt.so.13 => /a/libcublasLt.so.13'; + if (target.startsWith(soPath(fakeDirs.defaultDir))) + return 'libcublasLt.so.12 => /a/libcublasLt.so.12'; + } + throw new Error(`unexpected execFileSync(${cmd}, ${JSON.stringify(args)})`); + }, + }); + + expect(mod.getEffectiveOnnxRuntimeNodeDir()).toBe(fakeDirs.ourDir); + expect(mod.isEffectiveCudaAvailable()).toBe(true); + + mod.ensureOnnxRuntimeNodeMatchesSystem(); + // The real proof: registerHooks must actually fire. Before the toPosix() + // fix, the createRequire dispatcher's `from === ...` comparison would + // mismatch against a backslash-joined `from` under forceWin32Path, + // ensureOnnxRuntimeNodeMatchesSystem's outer try/catch would silently + // swallow the resulting MODULE_NOT_FOUND, and this would never fire. + expect(spy).toHaveBeenCalledTimes(1); + + const resolve = spy.mock.calls[0][0].resolve as ( + s: string, + c: never, + n: (s: string, c: never) => unknown, + ) => unknown; + const ctx = {} as never; + const next = vi.fn(); + const nodeResult = resolve('onnxruntime-node', ctx, next) as { + url: string; + shortCircuit: boolean; + }; + expect(nodeResult.shortCircuit).toBe(true); + expect(toPosix(nodeResult.url)).toContain('/fake/our/onnxruntime-node/index.js'); + expect(next).not.toHaveBeenCalled(); + }); +});