Fix building with CUDA < 12.8 (#2782)

2025-12-16 01:49:05 +08:00 · 2025-11-18 12:55:19 +09:00
parent 35f81728f1
commit 940f4c7818
40 changed files with 91 additions and 113 deletions
--- a/mlx/backend/cuda/gemms/cublas_gemm.cpp
+++ b/mlx/backend/cuda/gemms/cublas_gemm.cpp
@@ -370,7 +370,7 @@ void CublasGemm::execute(
    // Ensure workspace is 256-byte aligned
    int nbytes = cuda::ceil_div(heuristic_.workspaceSize, 256) * 256;
    array workspace(
-        cu::malloc_async(nbytes, encoder.stream()),
+        cu::malloc_async(nbytes, encoder),
        {static_cast<int>(heuristic_.workspaceSize)},
        int8);
    encoder.add_temporary(workspace);
--- a/mlx/backend/cuda/gemms/cublas_gemm_batched_12_9.cu
+++ b/mlx/backend/cuda/gemms/cublas_gemm_batched_12_9.cu
@@ -163,7 +163,7 @@ void CublasGemm::run_batched(

  // Launch kernel to set device offsets
  auto pointers = array(
-      cu::malloc_async(batch_count * sizeof(void*) * 3, encoder.stream()),
+      cu::malloc_async(batch_count * sizeof(void*) * 3, encoder),
      {batch_count * 3},
      uint64);

@@ -251,7 +251,7 @@ void CublasGemm::run_batched(

  // Launch kernel to set device offsets
  auto pointers = array(
-      cu::malloc_async(batch_count * sizeof(uint64_t) * 4, encoder.stream()),
+      cu::malloc_async(batch_count * sizeof(uint64_t) * 4, encoder),
      {batch_count * 4},
      uint64);