[CUDA] Reduce use of managed memory (#2725)

* Use async cuda malloc managed with cuda 13 * add pool threshold * refactor for regular cuda malloc * load eval gpu for cuda * remove use of cuda pool, use cuda free async * fix * fix * fix * fix * fix + comment
2025-12-16 01:49:05 +08:00 · 2025-11-05 16:05:23 -08:00
parent 27778156dc
commit df58b4133a
79 changed files with 795 additions and 515 deletions
--- a/mlx/backend/cuda/logsumexp.cu
+++ b/mlx/backend/cuda/logsumexp.cu
@@ -115,7 +115,7 @@ void LogSumExp::eval_gpu(const std::vector<array>& inputs, array& out) {

  auto in = ensure_contiguous(inputs[0]);
  if (in.flags().row_contiguous) {
-    out.set_data(allocator::malloc(out.nbytes()));
+    out.set_data(cu::malloc_async(out.nbytes(), encoder.stream()));
  } else {
    auto n = in.shape(-1);
    auto flags = in.flags();
@@ -130,7 +130,7 @@ void LogSumExp::eval_gpu(const std::vector<array>& inputs, array& out) {
    }
    flags.col_contiguous = col_contig;
    out.set_data(
-        allocator::malloc(in.nbytes() / n),
+        cu::malloc_async(in.nbytes() / n, encoder.stream()),
        in.data_size() / n,
        std::move(strides),
        flags);
@@ -151,8 +151,8 @@ void LogSumExp::eval_gpu(const std::vector<array>& inputs, array& out) {
          n_rows,
          block_dim(),
          0,
-          in.data<DataType>(),
-          out.data<DataType>(),
+          gpu_ptr<DataType>(in),
+          gpu_ptr<DataType>(out),
          axis_size);
    });
  });