This is an automated email from the ASF dual-hosted git repository.
ryankert01 pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/mahout.git
The following commit(s) were added to refs/heads/main by this push:
new 205ce75f6 fix(qdp-core): link cudart locally and stub FFI when toolkit
is absent (#1321)
205ce75f6 is described below
commit 205ce75f6b95c52ee02e14fdaa0f0ecd995c6cb9
Author: Andrew Musselman <[email protected]>
AuthorDate: Fri Jun 26 11:56:14 2026 -0700
fix(qdp-core): link cudart locally and stub FFI when toolkit is absent
(#1321)
* fix(qdp): own -lcudart linkage in qdp-core, gate stubs on qdp_no_cuda
* fix(qdp): check nvcc exit status and rerun build scripts on PATH
Address review feedback on the CUDA-detection probe shared by qdp-core
and qdp-kernels build scripts:
- output().is_ok() was true even when nvcc exited non-zero, so a
half-installed toolkit would set has_cuda = true and fall through to a
link error. Use .map(|o| o.status.success()).unwrap_or(false) in both
scripts so the two probes can't disagree.
- Add cargo:rerun-if-env-changed=PATH so installing the toolkit after a
stub build re-triggers detection without `cargo clean`.
Also run cargo fmt on the no_cuda_stubs module in cuda_ffi.rs to clear
the pre-commit fmt hook (was failing the test job).
* test(qdp): skip GPU tests when no CUDA device is available
This PR lets _qdp build and import without the CUDA toolkit (stub CUDA
Runtime symbols), so "extension importable" no longer implies a usable
GPU. The conftest auto-skip only checked extension availability, so on a
GPU-less runner the @pytest.mark.gpu tests ran against the stub engine
and aborted the pytest worker (SIGABRT) instead of being skipped.
- conftest: add a torch.cuda.is_available() probe and skip
@pytest.mark.gpu tests when no device is present, matching the inline
torch.cuda.is_available() guards already used across the QDP modules.
The existing "extension missing" skip behaviour is unchanged.
- mark test_synthetic_loader_batch_count @pytest.mark.gpu: it iterates a
synthetic loader, which encodes on the GPU.
Verified locally: with CUDA hidden the previously-crashing tests skip
cleanly (0 failures across testing/qdp + testing/qdp_python); with a GPU
present they run and pass.
---------
Co-authored-by: Hsien-Cheng Huang <[email protected]>
---
qdp/qdp-core/build.rs | 65 +++++++++++++-
qdp/qdp-core/src/gpu/cuda_ffi.rs | 117 +++++++++++++++++++++++++
qdp/qdp-kernels/build.rs | 10 ++-
testing/conftest.py | 60 ++++++++++---
testing/qdp_python/test_quantum_data_loader.py | 1 +
5 files changed, 238 insertions(+), 15 deletions(-)
diff --git a/qdp/qdp-core/build.rs b/qdp/qdp-core/build.rs
index 311ea139d..f90b3e979 100644
--- a/qdp/qdp-core/build.rs
+++ b/qdp/qdp-core/build.rs
@@ -14,10 +14,18 @@
// See the License for the specific language governing permissions and
// limitations under the License.
+use std::env;
+use std::process::Command;
+
fn main() {
+ compile_protos();
+ configure_cuda_linkage();
+}
+
+fn compile_protos() {
// Use vendored protoc to avoid missing protoc in CI/dev environments
unsafe {
- std::env::set_var("PROTOC",
protoc_bin_vendored::protoc_bin_path().unwrap());
+ env::set_var("PROTOC",
protoc_bin_vendored::protoc_bin_path().unwrap());
}
let mut config = prost_build::Config::new();
@@ -34,3 +42,58 @@ fn main() {
println!("cargo:rerun-if-changed=proto/tensor.proto");
}
+
+/// Detect the CUDA Runtime toolkit and emit the appropriate link directives.
+///
+/// `qdp-core` declares CUDA Runtime API extern symbols in
`src/gpu/cuda_ffi.rs`
+/// (cudaHostAlloc, cudaMemGetInfo, cudaEventCreateWithFlags, ...). Those
symbols
+/// must be resolved at link time, which requires `libcudart` from the CUDA
+/// Toolkit. Previously the only `-lcudart` directive lived in `qdp-kernels`'
+/// build script, and the `qdp_no_cuda` cfg it sets does not propagate
+/// cross-crate. The result was confusing linker errors on systems that have
+/// the NVIDIA driver but not the toolkit (e.g. PyTorch-only setups, where
+/// PyTorch ships its own bundled cudart inside the wheel).
+///
+/// This function:
+/// * emits `cargo:rustc-link-lib=cudart` and the appropriate
+/// `cargo:rustc-link-search` path when nvcc is found, and
+/// * emits `cargo:rustc-cfg=qdp_no_cuda` when it is not, gating the
+/// extern block in `cuda_ffi.rs` so the crate still links (with stubs
+/// that return a non-zero CUDA error at runtime).
+fn configure_cuda_linkage() {
+ println!("cargo::rustc-check-cfg=cfg(qdp_no_cuda)");
+ println!("cargo:rerun-if-env-changed=QDP_NO_CUDA");
+ println!("cargo:rerun-if-env-changed=CUDA_PATH");
+ // The whole CUDA-vs-stub decision hinges on finding nvcc on PATH, so a
PATH
+ // change (e.g. installing the toolkit) must re-trigger this script.
+ println!("cargo:rerun-if-env-changed=PATH");
+
+ let force_no_cuda = env::var("QDP_NO_CUDA")
+ .map(|v| v == "1" || v.eq_ignore_ascii_case("true") ||
v.eq_ignore_ascii_case("yes"))
+ .unwrap_or(false);
+
+ let has_cuda = !force_no_cuda
+ && Command::new("nvcc")
+ .arg("--version")
+ .output()
+ .map(|o| o.status.success())
+ .unwrap_or(false);
+
+ if !has_cuda {
+ println!("cargo:rustc-cfg=qdp_no_cuda");
+ println!(
+ "cargo:warning=qdp-core: CUDA toolkit not found (nvcc not in
PATH). \
+ Building with stub CUDA Runtime symbols; GPU functionality will
be \
+ unavailable at runtime. Install the CUDA Toolkit to enable GPU
support."
+ );
+ return;
+ }
+
+ let cuda_path = env::var("CUDA_PATH").unwrap_or_else(|_|
"/usr/local/cuda".to_string());
+ println!("cargo:rustc-link-search=native={}/lib64", cuda_path);
+ println!("cargo:rustc-link-lib=cudart");
+
+ // On macOS, also check /usr/local/cuda/lib
+ #[cfg(target_os = "macos")]
+ println!("cargo:rustc-link-search=native={}/lib", cuda_path);
+}
diff --git a/qdp/qdp-core/src/gpu/cuda_ffi.rs b/qdp/qdp-core/src/gpu/cuda_ffi.rs
index 2ed60c311..dc8930584 100644
--- a/qdp/qdp-core/src/gpu/cuda_ffi.rs
+++ b/qdp/qdp-core/src/gpu/cuda_ffi.rs
@@ -44,6 +44,19 @@ pub(crate) const CUDA_SUCCESS: i32 = 0;
#[allow(dead_code)]
pub(crate) const CUDA_ERROR_NOT_READY: i32 = 34;
+// CUDA Runtime API bindings.
+//
+// On systems with the CUDA Toolkit installed, `qdp-core`'s build script emits
+// `cargo:rustc-link-lib=cudart` and the extern block below is used unchanged.
+//
+// On systems without the toolkit (e.g. driver-only Linux hosts, or macOS),
+// the build script sets `qdp_no_cuda` and the stub block is used instead.
+// Stubs match each extern's signature and return a non-zero sentinel
+// (`QDP_CUDA_UNAVAILABLE`, mirroring qdp-kernels' "999" convention) so the
+// crate links cleanly but any actual GPU call surfaces as a runtime error
+// through the existing `if ret != 0 { Err(...) }` paths in the callers.
+
+#[cfg(not(qdp_no_cuda))]
unsafe extern "C" {
pub(crate) fn cudaHostAlloc(pHost: *mut *mut c_void, size: usize, flags:
u32) -> i32;
pub(crate) fn cudaFreeHost(ptr: *mut c_void) -> i32;
@@ -99,3 +112,107 @@ unsafe extern "C" {
/// Reference:
https://docs.nvidia.com/cuda/cuda-runtime-api/group__CUDART__EVENT.html
pub(crate) fn cudaEventElapsedTime(ms: *mut f32, start: *mut c_void, end:
*mut c_void) -> i32;
}
+
+// ---------------------------------------------------------------------------
+// Stub implementations when building without the CUDA toolkit (`qdp_no_cuda`).
+//
+// Wrapped in a private module so a single `#[allow(non_snake_case)]` covers
+// all stub names (the originals are camelCase to match the real CUDA Runtime
+// API; the `extern "C"` block above is exempt from this lint, but plain Rust
+// fns are not).
+// ---------------------------------------------------------------------------
+
+#[cfg(qdp_no_cuda)]
+#[allow(non_snake_case)]
+mod no_cuda_stubs {
+ use super::CudaPointerAttributes;
+ use std::ffi::c_void;
+
+ /// Sentinel error code returned by stub CUDA Runtime calls.
+ ///
+ /// Matches the "999" convention used by qdp-kernels' kernel-launcher
stubs.
+ const QDP_CUDA_UNAVAILABLE: i32 = 999;
+
+ pub(crate) unsafe fn cudaHostAlloc(_pHost: *mut *mut c_void, _size: usize,
_flags: u32) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaFreeHost(_ptr: *mut c_void) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ #[allow(dead_code)]
+ pub(crate) unsafe fn cudaPointerGetAttributes(
+ _attributes: *mut CudaPointerAttributes,
+ _ptr: *const c_void,
+ ) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaMemGetInfo(_free: *mut usize, _total: *mut usize)
-> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaMemcpyAsync(
+ _dst: *mut c_void,
+ _src: *const c_void,
+ _count: usize,
+ _kind: u32,
+ _stream: *mut c_void,
+ ) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaEventCreateWithFlags(_event: *mut *mut c_void,
_flags: u32) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaEventRecord(_event: *mut c_void, _stream: *mut
c_void) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaEventDestroy(_event: *mut c_void) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaStreamWaitEvent(
+ _stream: *mut c_void,
+ _event: *mut c_void,
+ _flags: u32,
+ ) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaStreamSynchronize(_stream: *mut c_void) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaMemsetAsync(
+ _devPtr: *mut c_void,
+ _value: i32,
+ _count: usize,
+ _stream: *mut c_void,
+ ) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ #[allow(dead_code)]
+ pub(crate) unsafe fn cudaEventQuery(_event: *mut c_void) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaEventSynchronize(_event: *mut c_void) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+
+ pub(crate) unsafe fn cudaEventElapsedTime(
+ _ms: *mut f32,
+ _start: *mut c_void,
+ _end: *mut c_void,
+ ) -> i32 {
+ QDP_CUDA_UNAVAILABLE
+ }
+}
+
+#[cfg(qdp_no_cuda)]
+pub(crate) use no_cuda_stubs::*;
diff --git a/qdp/qdp-kernels/build.rs b/qdp/qdp-kernels/build.rs
index def59d693..0dfeefdc0 100644
--- a/qdp/qdp-kernels/build.rs
+++ b/qdp/qdp-kernels/build.rs
@@ -167,6 +167,9 @@ fn main() {
println!("cargo:rerun-if-changed=src/phase.cu");
println!("cargo:rerun-if-env-changed=QDP_NO_CUDA");
println!("cargo:rerun-if-env-changed=QDP_CUDA_ARCH_LIST");
+ // The whole CUDA-vs-stub decision hinges on finding nvcc on PATH, so a
PATH
+ // change (e.g. installing the toolkit) must re-trigger this script.
+ println!("cargo:rerun-if-env-changed=PATH");
println!("cargo:rerun-if-changed=src/kernel_config.h");
// Check if CUDA is available by looking for nvcc
@@ -174,7 +177,12 @@ fn main() {
.map(|v| v == "1" || v.eq_ignore_ascii_case("true") ||
v.eq_ignore_ascii_case("yes"))
.unwrap_or(false);
- let has_cuda = !force_no_cuda &&
Command::new("nvcc").arg("--version").output().is_ok();
+ let has_cuda = !force_no_cuda
+ && Command::new("nvcc")
+ .arg("--version")
+ .output()
+ .map(|o| o.status.success())
+ .unwrap_or(false);
if !has_cuda {
// Expose a cfg for conditional compilation of stub symbols on Linux.
diff --git a/testing/conftest.py b/testing/conftest.py
index 28c1e150d..95009cc3c 100644
--- a/testing/conftest.py
+++ b/testing/conftest.py
@@ -40,6 +40,27 @@ except ImportError as e:
_QDP_IMPORT_ERROR = str(e)
+def _gpu_available() -> bool:
+ """Return True if a CUDA device is actually usable at runtime.
+
+ The ``_qdp`` extension can now be built and imported without the CUDA
+ toolkit -- it links stub CUDA Runtime symbols (see qdp-core ``build.rs`` /
+ ``cuda_ffi.rs``). So importing ``_qdp`` no longer implies a working GPU,
+ and ``@pytest.mark.gpu`` tests must additionally check for a real device.
+ This mirrors the ``torch.cuda.is_available()`` guard used by the inline
+ skips throughout the QDP test modules.
+ """
+ try:
+ import torch
+
+ return bool(torch.cuda.is_available())
+ except Exception:
+ return False
+
+
+_GPU_AVAILABLE = _gpu_available()
+
+
def pytest_configure(config):
"""Register custom pytest markers."""
config.addinivalue_line(
@@ -49,14 +70,23 @@ def pytest_configure(config):
def pytest_collection_modifyitems(config, items):
- """Auto-skip GPU/QDP tests if the _qdp extension is not available."""
- if _QDP_AVAILABLE:
- return
-
- skip_marker = pytest.mark.skip(
+ """Auto-skip QDP tests when the extension is missing, and GPU tests when
+ no CUDA device is available.
+
+ ``_qdp`` can now build/import without a GPU (stub CUDA symbols), so
+ extension availability alone no longer implies a usable device. GPU tests
+ are therefore skipped whenever ``torch.cuda.is_available()`` is False, even
+ when the extension imports -- otherwise they execute against the stub
+ runtime and abort the worker process.
+ """
+ skip_no_qdp = pytest.mark.skip(
reason=f"QDP extension not available: {_QDP_IMPORT_ERROR}. "
"Build with: cd qdp/qdp-python && maturin develop"
)
+ skip_no_gpu = pytest.mark.skip(
+ reason="GPU required: no CUDA device available. The _qdp extension is "
+ "importable but linked against CUDA stubs, or no GPU is present."
+ )
# Tests that work without _qdp (PyTorch reference backend tests).
_NO_QDP_OK = {
@@ -67,16 +97,20 @@ def pytest_collection_modifyitems(config, items):
}
for item in items:
- # Skip tests explicitly marked with @pytest.mark.gpu
- if "gpu" in item.keywords:
- item.add_marker(skip_marker)
+ is_gpu = "gpu" in item.keywords
- # Skip all tests in testing/qdp/ directory, except those that
- # explicitly do not require the Rust extension.
fspath_str = str(item.fspath)
- if "testing/qdp" in fspath_str or "testing\\qdp" in fspath_str:
- if not any(name in fspath_str for name in _NO_QDP_OK):
- item.add_marker(skip_marker)
+ needs_qdp = (
+ "testing/qdp" in fspath_str or "testing\\qdp" in fspath_str
+ ) and not any(name in fspath_str for name in _NO_QDP_OK)
+
+ if not _QDP_AVAILABLE:
+ # No extension at all: skip GPU tests and everything needing _qdp.
+ if is_gpu or needs_qdp:
+ item.add_marker(skip_no_qdp)
+ elif is_gpu and not _GPU_AVAILABLE:
+ # Extension built (possibly with CUDA stubs) but no usable device.
+ item.add_marker(skip_no_gpu)
@pytest.fixture
diff --git a/testing/qdp_python/test_quantum_data_loader.py
b/testing/qdp_python/test_quantum_data_loader.py
index 283d67322..9dcec2324 100644
--- a/testing/qdp_python/test_quantum_data_loader.py
+++ b/testing/qdp_python/test_quantum_data_loader.py
@@ -92,6 +92,7 @@ def test_source_file_empty_path_raises(loader_cls:
type[QuantumDataLoaderType]):
@pytest.mark.skipif(not _loader_available(), reason="QuantumDataLoader not
available")
[email protected]
def test_synthetic_loader_batch_count(loader_cls: type[QuantumDataLoaderType]):
"""Synthetic loader yields exactly total_batches batches."""
total = 5