[CUDA] Reduce use of managed memory (#2725)

* Use async cuda malloc managed with cuda 13 * add pool threshold * refactor for regular cuda malloc * load eval gpu for cuda * remove use of cuda pool, use cuda free async * fix * fix * fix * fix * fix + comment
2025-12-16 01:49:05 +08:00 · 2025-11-05 16:05:23 -08:00
parent 27778156dc
commit df58b4133a
79 changed files with 795 additions and 515 deletions
--- a/mlx/backend/cuda/sort.cu
+++ b/mlx/backend/cuda/sort.cu
@@ -49,11 +49,14 @@ void gpu_sort(const Stream& s, array in, array& out_, int axis, bool argsort) {
    array trans = swapaxes_in_eval(in, axis, last_dim);
    in = contiguous_copy_gpu(trans, s);
    encoder.add_temporary(in);
-    out = array(allocator::malloc(out.nbytes()), in.shape(), out.dtype());
+    out = array(
+        cu::malloc_async(out.nbytes(), encoder.stream()),
+        in.shape(),
+        out.dtype());
    encoder.add_temporary(out);
  } else {
    out.set_data(
-        allocator::malloc(in.data_size() * out.itemsize()),
+        cu::malloc_async(in.data_size() * out.itemsize(), encoder.stream()),
        in.data_size(),
        in.strides(),
        in.flags());
@@ -70,22 +73,28 @@ void gpu_sort(const Stream& s, array in, array& out_, int axis, bool argsort) {
          thrust::make_counting_iterator(0), OffsetTransform{nsort});
      if (argsort) {
        // Indices in the sorted dimension.
-        array indices(allocator::malloc(out.nbytes()), in.shape(), out.dtype());
+        array indices(
+            cu::malloc_async(out.nbytes(), encoder.stream()),
+            in.shape(),
+            out.dtype());
        encoder.add_temporary(indices);

        // In argsort though we don't need the result of sorted values, the
        // API requires us to provide an array to store it.
-        array discard(allocator::malloc(in.nbytes()), in.shape(), in.dtype());
+        array discard(
+            cu::malloc_async(in.nbytes(), encoder.stream()),
+            in.shape(),
+            in.dtype());
        encoder.add_temporary(discard);

        size_t size;
        CHECK_CUDA_ERROR(cub::DeviceSegmentedRadixSort::SortPairs(
            nullptr,
            size,
-            in.data<Type>(),
-            discard.data<Type>(),
-            indices.data<uint32_t>(),
-            out.data<uint32_t>(),
+            gpu_ptr<Type>(in),
+            gpu_ptr<Type>(discard),
+            gpu_ptr<uint32_t>(indices),
+            gpu_ptr<uint32_t>(out),
            in.data_size(),
            in.data_size() / nsort,
            offsets,
@@ -94,7 +103,10 @@ void gpu_sort(const Stream& s, array in, array& out_, int axis, bool argsort) {
            sizeof(Type) * 8,
            stream));

-        array temp(allocator::malloc(size), {static_cast<int>(size)}, uint8);
+        array temp(
+            cu::malloc_async(size, encoder.stream()),
+            {static_cast<int>(size)},
+            uint8);
        encoder.add_temporary(temp);

        // Start capturing after allocations
@@ -103,16 +115,16 @@ void gpu_sort(const Stream& s, array in, array& out_, int axis, bool argsort) {
            cu::thrust_policy(stream),
            thrust::counting_iterator<uint32_t>(0),
            thrust::counting_iterator<uint32_t>(indices.data_size()),
-            thrust::device_pointer_cast(indices.data<uint32_t>()),
+            thrust::device_pointer_cast(gpu_ptr<uint32_t>(indices)),
            ModOp<uint32_t>{static_cast<uint32_t>(nsort)});

        CHECK_CUDA_ERROR(cub::DeviceSegmentedRadixSort::SortPairs(
-            temp.data<void>(),
+            gpu_ptr<void>(temp),
            size,
-            in.data<Type>(),
-            discard.data<Type>(),
-            indices.data<uint32_t>(),
-            out.data<uint32_t>(),
+            gpu_ptr<Type>(in),
+            gpu_ptr<Type>(discard),
+            gpu_ptr<uint32_t>(indices),
+            gpu_ptr<uint32_t>(out),
            in.data_size(),
            in.data_size() / nsort,
            offsets,
@@ -125,8 +137,8 @@ void gpu_sort(const Stream& s, array in, array& out_, int axis, bool argsort) {
        CHECK_CUDA_ERROR(cub::DeviceSegmentedRadixSort::SortKeys(
            nullptr,
            size,
-            in.data<Type>(),
-            out.data<Type>(),
+            gpu_ptr<Type>(in),
+            gpu_ptr<Type>(out),
            in.data_size(),
            in.data_size() / nsort,
            offsets,
@@ -135,16 +147,19 @@ void gpu_sort(const Stream& s, array in, array& out_, int axis, bool argsort) {
            sizeof(Type) * 8,
            stream));

-        array temp(allocator::malloc(size), {static_cast<int>(size)}, uint8);
+        array temp(
+            cu::malloc_async(size, encoder.stream()),
+            {static_cast<int>(size)},
+            uint8);
        encoder.add_temporary(temp);

        // Start capturing after allocations
        auto capture = encoder.capture_context();
        CHECK_CUDA_ERROR(cub::DeviceSegmentedRadixSort::SortKeys(
-            temp.data<void>(),
+            gpu_ptr<void>(temp),
            size,
-            in.data<Type>(),
-            out.data<Type>(),
+            gpu_ptr<Type>(in),
+            gpu_ptr<Type>(out),
            in.data_size(),
            in.data_size() / nsort,
            offsets,