This is an automated email from the ASF dual-hosted git repository.

ryankert01 pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/mahout.git


The following commit(s) were added to refs/heads/main by this push:
     new 205ce75f6 fix(qdp-core): link cudart locally and stub FFI when toolkit 
is absent  (#1321)
205ce75f6 is described below

commit 205ce75f6b95c52ee02e14fdaa0f0ecd995c6cb9
Author: Andrew Musselman <[email protected]>
AuthorDate: Fri Jun 26 11:56:14 2026 -0700

    fix(qdp-core): link cudart locally and stub FFI when toolkit is absent  
(#1321)
    
    * fix(qdp): own -lcudart linkage in qdp-core, gate stubs on qdp_no_cuda
    
    * fix(qdp): check nvcc exit status and rerun build scripts on PATH
    
    Address review feedback on the CUDA-detection probe shared by qdp-core
    and qdp-kernels build scripts:
    
    - output().is_ok() was true even when nvcc exited non-zero, so a
      half-installed toolkit would set has_cuda = true and fall through to a
      link error. Use .map(|o| o.status.success()).unwrap_or(false) in both
      scripts so the two probes can't disagree.
    - Add cargo:rerun-if-env-changed=PATH so installing the toolkit after a
      stub build re-triggers detection without `cargo clean`.
    
    Also run cargo fmt on the no_cuda_stubs module in cuda_ffi.rs to clear
    the pre-commit fmt hook (was failing the test job).
    
    * test(qdp): skip GPU tests when no CUDA device is available
    
    This PR lets _qdp build and import without the CUDA toolkit (stub CUDA
    Runtime symbols), so "extension importable" no longer implies a usable
    GPU. The conftest auto-skip only checked extension availability, so on a
    GPU-less runner the @pytest.mark.gpu tests ran against the stub engine
    and aborted the pytest worker (SIGABRT) instead of being skipped.
    
    - conftest: add a torch.cuda.is_available() probe and skip
      @pytest.mark.gpu tests when no device is present, matching the inline
      torch.cuda.is_available() guards already used across the QDP modules.
      The existing "extension missing" skip behaviour is unchanged.
    - mark test_synthetic_loader_batch_count @pytest.mark.gpu: it iterates a
      synthetic loader, which encodes on the GPU.
    
    Verified locally: with CUDA hidden the previously-crashing tests skip
    cleanly (0 failures across testing/qdp + testing/qdp_python); with a GPU
    present they run and pass.
    
    ---------
    
    Co-authored-by: Hsien-Cheng Huang <[email protected]>
---
 qdp/qdp-core/build.rs                          |  65 +++++++++++++-
 qdp/qdp-core/src/gpu/cuda_ffi.rs               | 117 +++++++++++++++++++++++++
 qdp/qdp-kernels/build.rs                       |  10 ++-
 testing/conftest.py                            |  60 ++++++++++---
 testing/qdp_python/test_quantum_data_loader.py |   1 +
 5 files changed, 238 insertions(+), 15 deletions(-)

diff --git a/qdp/qdp-core/build.rs b/qdp/qdp-core/build.rs
index 311ea139d..f90b3e979 100644
--- a/qdp/qdp-core/build.rs
+++ b/qdp/qdp-core/build.rs
@@ -14,10 +14,18 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.
 
+use std::env;
+use std::process::Command;
+
 fn main() {
+    compile_protos();
+    configure_cuda_linkage();
+}
+
+fn compile_protos() {
     // Use vendored protoc to avoid missing protoc in CI/dev environments
     unsafe {
-        std::env::set_var("PROTOC", 
protoc_bin_vendored::protoc_bin_path().unwrap());
+        env::set_var("PROTOC", 
protoc_bin_vendored::protoc_bin_path().unwrap());
     }
 
     let mut config = prost_build::Config::new();
@@ -34,3 +42,58 @@ fn main() {
 
     println!("cargo:rerun-if-changed=proto/tensor.proto");
 }
+
+/// Detect the CUDA Runtime toolkit and emit the appropriate link directives.
+///
+/// `qdp-core` declares CUDA Runtime API extern symbols in 
`src/gpu/cuda_ffi.rs`
+/// (cudaHostAlloc, cudaMemGetInfo, cudaEventCreateWithFlags, ...). Those 
symbols
+/// must be resolved at link time, which requires `libcudart` from the CUDA
+/// Toolkit. Previously the only `-lcudart` directive lived in `qdp-kernels`'
+/// build script, and the `qdp_no_cuda` cfg it sets does not propagate
+/// cross-crate. The result was confusing linker errors on systems that have
+/// the NVIDIA driver but not the toolkit (e.g. PyTorch-only setups, where
+/// PyTorch ships its own bundled cudart inside the wheel).
+///
+/// This function:
+///   * emits `cargo:rustc-link-lib=cudart` and the appropriate
+///     `cargo:rustc-link-search` path when nvcc is found, and
+///   * emits `cargo:rustc-cfg=qdp_no_cuda` when it is not, gating the
+///     extern block in `cuda_ffi.rs` so the crate still links (with stubs
+///     that return a non-zero CUDA error at runtime).
+fn configure_cuda_linkage() {
+    println!("cargo::rustc-check-cfg=cfg(qdp_no_cuda)");
+    println!("cargo:rerun-if-env-changed=QDP_NO_CUDA");
+    println!("cargo:rerun-if-env-changed=CUDA_PATH");
+    // The whole CUDA-vs-stub decision hinges on finding nvcc on PATH, so a 
PATH
+    // change (e.g. installing the toolkit) must re-trigger this script.
+    println!("cargo:rerun-if-env-changed=PATH");
+
+    let force_no_cuda = env::var("QDP_NO_CUDA")
+        .map(|v| v == "1" || v.eq_ignore_ascii_case("true") || 
v.eq_ignore_ascii_case("yes"))
+        .unwrap_or(false);
+
+    let has_cuda = !force_no_cuda
+        && Command::new("nvcc")
+            .arg("--version")
+            .output()
+            .map(|o| o.status.success())
+            .unwrap_or(false);
+
+    if !has_cuda {
+        println!("cargo:rustc-cfg=qdp_no_cuda");
+        println!(
+            "cargo:warning=qdp-core: CUDA toolkit not found (nvcc not in 
PATH). \
+             Building with stub CUDA Runtime symbols; GPU functionality will 
be \
+             unavailable at runtime. Install the CUDA Toolkit to enable GPU 
support."
+        );
+        return;
+    }
+
+    let cuda_path = env::var("CUDA_PATH").unwrap_or_else(|_| 
"/usr/local/cuda".to_string());
+    println!("cargo:rustc-link-search=native={}/lib64", cuda_path);
+    println!("cargo:rustc-link-lib=cudart");
+
+    // On macOS, also check /usr/local/cuda/lib
+    #[cfg(target_os = "macos")]
+    println!("cargo:rustc-link-search=native={}/lib", cuda_path);
+}
diff --git a/qdp/qdp-core/src/gpu/cuda_ffi.rs b/qdp/qdp-core/src/gpu/cuda_ffi.rs
index 2ed60c311..dc8930584 100644
--- a/qdp/qdp-core/src/gpu/cuda_ffi.rs
+++ b/qdp/qdp-core/src/gpu/cuda_ffi.rs
@@ -44,6 +44,19 @@ pub(crate) const CUDA_SUCCESS: i32 = 0;
 #[allow(dead_code)]
 pub(crate) const CUDA_ERROR_NOT_READY: i32 = 34;
 
+// CUDA Runtime API bindings.
+//
+// On systems with the CUDA Toolkit installed, `qdp-core`'s build script emits
+// `cargo:rustc-link-lib=cudart` and the extern block below is used unchanged.
+//
+// On systems without the toolkit (e.g. driver-only Linux hosts, or macOS),
+// the build script sets `qdp_no_cuda` and the stub block is used instead.
+// Stubs match each extern's signature and return a non-zero sentinel
+// (`QDP_CUDA_UNAVAILABLE`, mirroring qdp-kernels' "999" convention) so the
+// crate links cleanly but any actual GPU call surfaces as a runtime error
+// through the existing `if ret != 0 { Err(...) }` paths in the callers.
+
+#[cfg(not(qdp_no_cuda))]
 unsafe extern "C" {
     pub(crate) fn cudaHostAlloc(pHost: *mut *mut c_void, size: usize, flags: 
u32) -> i32;
     pub(crate) fn cudaFreeHost(ptr: *mut c_void) -> i32;
@@ -99,3 +112,107 @@ unsafe extern "C" {
     /// Reference: 
https://docs.nvidia.com/cuda/cuda-runtime-api/group__CUDART__EVENT.html
     pub(crate) fn cudaEventElapsedTime(ms: *mut f32, start: *mut c_void, end: 
*mut c_void) -> i32;
 }
+
+// ---------------------------------------------------------------------------
+// Stub implementations when building without the CUDA toolkit (`qdp_no_cuda`).
+//
+// Wrapped in a private module so a single `#[allow(non_snake_case)]` covers
+// all stub names (the originals are camelCase to match the real CUDA Runtime
+// API; the `extern "C"` block above is exempt from this lint, but plain Rust
+// fns are not).
+// ---------------------------------------------------------------------------
+
+#[cfg(qdp_no_cuda)]
+#[allow(non_snake_case)]
+mod no_cuda_stubs {
+    use super::CudaPointerAttributes;
+    use std::ffi::c_void;
+
+    /// Sentinel error code returned by stub CUDA Runtime calls.
+    ///
+    /// Matches the "999" convention used by qdp-kernels' kernel-launcher 
stubs.
+    const QDP_CUDA_UNAVAILABLE: i32 = 999;
+
+    pub(crate) unsafe fn cudaHostAlloc(_pHost: *mut *mut c_void, _size: usize, 
_flags: u32) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaFreeHost(_ptr: *mut c_void) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    #[allow(dead_code)]
+    pub(crate) unsafe fn cudaPointerGetAttributes(
+        _attributes: *mut CudaPointerAttributes,
+        _ptr: *const c_void,
+    ) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaMemGetInfo(_free: *mut usize, _total: *mut usize) 
-> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaMemcpyAsync(
+        _dst: *mut c_void,
+        _src: *const c_void,
+        _count: usize,
+        _kind: u32,
+        _stream: *mut c_void,
+    ) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaEventCreateWithFlags(_event: *mut *mut c_void, 
_flags: u32) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaEventRecord(_event: *mut c_void, _stream: *mut 
c_void) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaEventDestroy(_event: *mut c_void) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaStreamWaitEvent(
+        _stream: *mut c_void,
+        _event: *mut c_void,
+        _flags: u32,
+    ) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaStreamSynchronize(_stream: *mut c_void) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaMemsetAsync(
+        _devPtr: *mut c_void,
+        _value: i32,
+        _count: usize,
+        _stream: *mut c_void,
+    ) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    #[allow(dead_code)]
+    pub(crate) unsafe fn cudaEventQuery(_event: *mut c_void) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaEventSynchronize(_event: *mut c_void) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+
+    pub(crate) unsafe fn cudaEventElapsedTime(
+        _ms: *mut f32,
+        _start: *mut c_void,
+        _end: *mut c_void,
+    ) -> i32 {
+        QDP_CUDA_UNAVAILABLE
+    }
+}
+
+#[cfg(qdp_no_cuda)]
+pub(crate) use no_cuda_stubs::*;
diff --git a/qdp/qdp-kernels/build.rs b/qdp/qdp-kernels/build.rs
index def59d693..0dfeefdc0 100644
--- a/qdp/qdp-kernels/build.rs
+++ b/qdp/qdp-kernels/build.rs
@@ -167,6 +167,9 @@ fn main() {
     println!("cargo:rerun-if-changed=src/phase.cu");
     println!("cargo:rerun-if-env-changed=QDP_NO_CUDA");
     println!("cargo:rerun-if-env-changed=QDP_CUDA_ARCH_LIST");
+    // The whole CUDA-vs-stub decision hinges on finding nvcc on PATH, so a 
PATH
+    // change (e.g. installing the toolkit) must re-trigger this script.
+    println!("cargo:rerun-if-env-changed=PATH");
     println!("cargo:rerun-if-changed=src/kernel_config.h");
 
     // Check if CUDA is available by looking for nvcc
@@ -174,7 +177,12 @@ fn main() {
         .map(|v| v == "1" || v.eq_ignore_ascii_case("true") || 
v.eq_ignore_ascii_case("yes"))
         .unwrap_or(false);
 
-    let has_cuda = !force_no_cuda && 
Command::new("nvcc").arg("--version").output().is_ok();
+    let has_cuda = !force_no_cuda
+        && Command::new("nvcc")
+            .arg("--version")
+            .output()
+            .map(|o| o.status.success())
+            .unwrap_or(false);
 
     if !has_cuda {
         // Expose a cfg for conditional compilation of stub symbols on Linux.
diff --git a/testing/conftest.py b/testing/conftest.py
index 28c1e150d..95009cc3c 100644
--- a/testing/conftest.py
+++ b/testing/conftest.py
@@ -40,6 +40,27 @@ except ImportError as e:
     _QDP_IMPORT_ERROR = str(e)
 
 
+def _gpu_available() -> bool:
+    """Return True if a CUDA device is actually usable at runtime.
+
+    The ``_qdp`` extension can now be built and imported without the CUDA
+    toolkit -- it links stub CUDA Runtime symbols (see qdp-core ``build.rs`` /
+    ``cuda_ffi.rs``).  So importing ``_qdp`` no longer implies a working GPU,
+    and ``@pytest.mark.gpu`` tests must additionally check for a real device.
+    This mirrors the ``torch.cuda.is_available()`` guard used by the inline
+    skips throughout the QDP test modules.
+    """
+    try:
+        import torch
+
+        return bool(torch.cuda.is_available())
+    except Exception:
+        return False
+
+
+_GPU_AVAILABLE = _gpu_available()
+
+
 def pytest_configure(config):
     """Register custom pytest markers."""
     config.addinivalue_line(
@@ -49,14 +70,23 @@ def pytest_configure(config):
 
 
 def pytest_collection_modifyitems(config, items):
-    """Auto-skip GPU/QDP tests if the _qdp extension is not available."""
-    if _QDP_AVAILABLE:
-        return
-
-    skip_marker = pytest.mark.skip(
+    """Auto-skip QDP tests when the extension is missing, and GPU tests when
+    no CUDA device is available.
+
+    ``_qdp`` can now build/import without a GPU (stub CUDA symbols), so
+    extension availability alone no longer implies a usable device.  GPU tests
+    are therefore skipped whenever ``torch.cuda.is_available()`` is False, even
+    when the extension imports -- otherwise they execute against the stub
+    runtime and abort the worker process.
+    """
+    skip_no_qdp = pytest.mark.skip(
         reason=f"QDP extension not available: {_QDP_IMPORT_ERROR}. "
         "Build with: cd qdp/qdp-python && maturin develop"
     )
+    skip_no_gpu = pytest.mark.skip(
+        reason="GPU required: no CUDA device available. The _qdp extension is "
+        "importable but linked against CUDA stubs, or no GPU is present."
+    )
 
     # Tests that work without _qdp (PyTorch reference backend tests).
     _NO_QDP_OK = {
@@ -67,16 +97,20 @@ def pytest_collection_modifyitems(config, items):
     }
 
     for item in items:
-        # Skip tests explicitly marked with @pytest.mark.gpu
-        if "gpu" in item.keywords:
-            item.add_marker(skip_marker)
+        is_gpu = "gpu" in item.keywords
 
-        # Skip all tests in testing/qdp/ directory, except those that
-        # explicitly do not require the Rust extension.
         fspath_str = str(item.fspath)
-        if "testing/qdp" in fspath_str or "testing\\qdp" in fspath_str:
-            if not any(name in fspath_str for name in _NO_QDP_OK):
-                item.add_marker(skip_marker)
+        needs_qdp = (
+            "testing/qdp" in fspath_str or "testing\\qdp" in fspath_str
+        ) and not any(name in fspath_str for name in _NO_QDP_OK)
+
+        if not _QDP_AVAILABLE:
+            # No extension at all: skip GPU tests and everything needing _qdp.
+            if is_gpu or needs_qdp:
+                item.add_marker(skip_no_qdp)
+        elif is_gpu and not _GPU_AVAILABLE:
+            # Extension built (possibly with CUDA stubs) but no usable device.
+            item.add_marker(skip_no_gpu)
 
 
 @pytest.fixture
diff --git a/testing/qdp_python/test_quantum_data_loader.py 
b/testing/qdp_python/test_quantum_data_loader.py
index 283d67322..9dcec2324 100644
--- a/testing/qdp_python/test_quantum_data_loader.py
+++ b/testing/qdp_python/test_quantum_data_loader.py
@@ -92,6 +92,7 @@ def test_source_file_empty_path_raises(loader_cls: 
type[QuantumDataLoaderType]):
 
 
 @pytest.mark.skipif(not _loader_available(), reason="QuantumDataLoader not 
available")
[email protected]
 def test_synthetic_loader_batch_count(loader_cls: type[QuantumDataLoaderType]):
     """Synthetic loader yields exactly total_batches batches."""
     total = 5

Reply via email to