[CUDA] Reduce use of managed memory (#2725)

* Use async cuda malloc managed with cuda 13 * add pool threshold * refactor for regular cuda malloc * load eval gpu for cuda * remove use of cuda pool, use cuda free async * fix * fix * fix * fix * fix + comment
2025-12-16 01:49:05 +08:00 · 2025-11-05 16:05:23 -08:00
parent 27778156dc
commit df58b4133a
79 changed files with 795 additions and 515 deletions
--- a/mlx/backend/cuda/fence.cpp
+++ b/mlx/backend/cuda/fence.cpp
@@ -1,6 +1,8 @@
 // Copyright © 2025 Apple Inc.

 #include "mlx/fence.h"
+#include "mlx/backend/cuda/allocator.h"
+#include "mlx/backend/cuda/device.h"
 #include "mlx/backend/cuda/event.h"

 namespace mlx::core {
@@ -20,8 +22,24 @@ void Fence::wait(Stream s, const array&) {
  fence->event.wait(fence->count);
 }

-void Fence::update(Stream s, const array&) {
+void Fence::update(Stream s, const array& a, bool cross_device) {
  auto* fence = static_cast<FenceImpl*>(fence_.get());
+  if (cross_device) {
+    // Move to managed memory if there is a device switch
+    auto& cbuf =
+        *static_cast<cu::CudaBuffer*>(const_cast<array&>(a).buffer().ptr());
+    if (cbuf.device != -1) {
+      void* new_data;
+      CHECK_CUDA_ERROR(cudaMallocManaged(&new_data, cbuf.size));
+      cbuf.device = -1;
+      auto& encoder = cu::device(s.device).get_command_encoder(s);
+      encoder.commit();
+      CHECK_CUDA_ERROR(cudaMemcpyAsync(
+          new_data, cbuf.data, cbuf.size, cudaMemcpyDefault, encoder.stream()));
+      CHECK_CUDA_ERROR(cudaFreeAsync(cbuf.data, encoder.stream()));
+      cbuf.data = new_data;
+    }
+  }
  fence->count++;
  fence->event.signal(s, fence->count);
 }