[CUDA] Reduce use of managed memory (#2725)

* Use async cuda malloc managed with cuda 13 * add pool threshold * refactor for regular cuda malloc * load eval gpu for cuda * remove use of cuda pool, use cuda free async * fix * fix * fix * fix * fix + comment
2025-12-16 01:49:05 +08:00 · 2025-11-05 16:05:23 -08:00
parent 27778156dc
commit df58b4133a
79 changed files with 795 additions and 515 deletions
--- a/mlx/backend/cuda/gemms/cublas_gemm_batched_12_0.cpp
+++ b/mlx/backend/cuda/gemms/cublas_gemm_batched_12_0.cpp
@@ -25,9 +25,10 @@ void CublasGemm::run_batched(
  for (size_t i = 0; i < nbatch; ++i) {
    execute(
        encoder,
-        out.data<int8_t>() + out.itemsize() * i * batch_shape.back() * M_ * N_,
-        a.data<int8_t>() + a.itemsize() * a_it.loc,
-        b.data<int8_t>() + b.itemsize() * b_it.loc,
+        gpu_ptr<int8_t>(out) +
+            out.itemsize() * i * batch_shape.back() * M_ * N_,
+        gpu_ptr<int8_t>(a) + a.itemsize() * a_it.loc,
+        gpu_ptr<int8_t>(b) + b.itemsize() * b_it.loc,
        nullptr,
        alpha);
    a_it.step();
@@ -60,10 +61,11 @@ void CublasGemm::run_batched(
  for (size_t i = 0; i < nbatch; ++i) {
    execute(
        encoder,
-        out.data<int8_t>() + out.itemsize() * i * batch_shape.back() * M_ * N_,
-        a.data<int8_t>() + a.itemsize() * a_it.loc,
-        b.data<int8_t>() + b.itemsize() * b_it.loc,
-        c.data<int8_t>() + c.itemsize() * c_it.loc,
+        gpu_ptr<int8_t>(out) +
+            out.itemsize() * i * batch_shape.back() * M_ * N_,
+        gpu_ptr<int8_t>(a) + a.itemsize() * a_it.loc,
+        gpu_ptr<int8_t>(b) + b.itemsize() * b_it.loc,
+        gpu_ptr<int8_t>(c) + c.itemsize() * c_it.loc,
        alpha,
        beta);
    a_it.step();