[CUDA] Reduce use of managed memory (#2725)

* Use async cuda malloc managed with cuda 13 * add pool threshold * refactor for regular cuda malloc * load eval gpu for cuda * remove use of cuda pool, use cuda free async * fix * fix * fix * fix * fix + comment
2025-12-16 01:49:05 +08:00 · 2025-11-05 16:05:23 -08:00
parent 27778156dc
commit df58b4133a
79 changed files with 795 additions and 515 deletions
--- a/mlx/backend/cuda/quantized/affine_quantize.cu
+++ b/mlx/backend/cuda/quantized/affine_quantize.cu
@@ -262,10 +262,10 @@ void affine_quantize(
            num_blocks,
            block_dims,
            0,
-            w.data<T>(),
-            wq.data<uint8_t>(),
-            scales.data<T>(),
-            biases.data<T>(),
+            gpu_ptr<T>(w),
+            gpu_ptr<uint8_t>(wq),
+            gpu_ptr<T>(scales),
+            gpu_ptr<T>(biases),
            w.size());
      });
    });
@@ -318,10 +318,10 @@ void affine_dequantize(
            num_blocks,
            block_dims,
            0,
-            wq.data<uint8_t>(),
-            scales.data<T>(),
-            biases.data<T>(),
-            w.data<T>(),
+            gpu_ptr<uint8_t>(wq),
+            gpu_ptr<T>(scales),
+            gpu_ptr<T>(biases),
+            gpu_ptr<T>(w),
            w.size());
      });
    });