redesign for faster cpu/gpu synch (#1869)

* redesign for faster cpu/gpu synch * load + more async CPU * use command encoder API and move more ops to use it * make fence back-end generic + CPU only fence * faster build * fix async eval * fixes + handle temporaries * fix / improve cpu conv * remove unused status, fix siblings * fix extensions * fix * fix no cpu build * format * comments * fix perf regression, remove unecessary abort * fix events, task limit cpu * fix waiting * fix donation / temporaries in normalization
2025-12-16 01:49:05 +08:00 · 2025-03-06 19:23:38 -08:00
parent 5245f12a46
commit c4230747a1
103 changed files with 5013 additions and 3873 deletions
--- a/mlx/backend/common/binary.h
+++ b/mlx/backend/common/binary.h
@@ -38,8 +38,7 @@ inline void set_binary_op_output_data(
    const array& a,
    const array& b,
    array& out,
-    BinaryOpType bopt,
-    bool donate_with_move = false) {
+    BinaryOpType bopt) {
  bool b_donatable = is_donatable(b, out);
  bool a_donatable = is_donatable(a, out);
  switch (bopt) {
@@ -49,11 +48,7 @@ inline void set_binary_op_output_data(
      break;
    case BinaryOpType::ScalarVector:
      if (b_donatable) {
-        if (donate_with_move) {
-          out.move_shared_buffer(b);
-        } else {
-          out.copy_shared_buffer(b);
-        }
+        out.copy_shared_buffer(b);
      } else {
        out.set_data(
            allocator::malloc_or_wait(b.data_size() * out.itemsize()),
@@ -64,11 +59,7 @@ inline void set_binary_op_output_data(
      break;
    case BinaryOpType::VectorScalar:
      if (a_donatable) {
-        if (donate_with_move) {
-          out.move_shared_buffer(a);
-        } else {
-          out.copy_shared_buffer(a);
-        }
+        out.copy_shared_buffer(a);
      } else {
        out.set_data(
            allocator::malloc_or_wait(a.data_size() * out.itemsize()),
@@ -79,17 +70,9 @@ inline void set_binary_op_output_data(
      break;
    case BinaryOpType::VectorVector:
      if (a_donatable) {
-        if (donate_with_move) {
-          out.move_shared_buffer(a);
-        } else {
-          out.copy_shared_buffer(a);
-        }
+        out.copy_shared_buffer(a);
      } else if (b_donatable) {
-        if (donate_with_move) {
-          out.move_shared_buffer(b);
-        } else {
-          out.copy_shared_buffer(b);
-        }
+        out.copy_shared_buffer(b);
      } else {
        out.set_data(
            allocator::malloc_or_wait(a.data_size() * out.itemsize()),
@@ -100,18 +83,10 @@ inline void set_binary_op_output_data(
      break;
    case BinaryOpType::General:
      if (a_donatable && a.flags().row_contiguous && a.size() == out.size()) {
-        if (donate_with_move) {
-          out.move_shared_buffer(a);
-        } else {
-          out.copy_shared_buffer(a);
-        }
+        out.copy_shared_buffer(a);
      } else if (
          b_donatable && b.flags().row_contiguous && b.size() == out.size()) {
-        if (donate_with_move) {
-          out.move_shared_buffer(b);
-        } else {
-          out.copy_shared_buffer(b);
-        }
+        out.copy_shared_buffer(b);
      } else {
        out.set_data(allocator::malloc_or_wait(out.nbytes()));
      }
--- a/mlx/backend/common/common.cpp
+++ b/mlx/backend/common/common.cpp
@@ -39,7 +39,7 @@ void AsStrided::eval(const std::vector<array>& inputs, array& out) {
  // rely on data_size anyway.
  size_t data_size = out.size();

-  return move_or_copy(in, out, strides_, flags, data_size, offset_);
+  return out.copy_shared_buffer(in, strides_, flags, data_size, offset_);
 }

 void broadcast(const array& in, array& out) {
@@ -56,7 +56,7 @@ void broadcast(const array& in, array& out) {
  if (out.size() > in.size()) {
    flags.row_contiguous = flags.col_contiguous = false;
  }
-  move_or_copy(in, out, strides, flags, in.data_size());
+  out.copy_shared_buffer(in, strides, flags, in.data_size());
 }

 void Broadcast::eval(const std::vector<array>& inputs, array& out) {
@@ -69,7 +69,7 @@ void BroadcastAxes::eval(const std::vector<array>& inputs, array& out) {

 void Copy::eval(const std::vector<array>& inputs, array& out) {
  assert(inputs.size() == 1);
-  move_or_copy(inputs[0], out);
+  out.copy_shared_buffer(inputs[0]);
 }

 void CustomTransforms::eval(
@@ -78,7 +78,7 @@ void CustomTransforms::eval(
  assert(inputs.size() > outputs.size());
  for (int i = 0, j = inputs.size() - outputs.size(); i < outputs.size();
       i++, j++) {
-    move_or_copy(inputs[j], outputs[i]);
+    outputs[i].copy_shared_buffer(inputs[j]);
  }
 }

@@ -87,7 +87,7 @@ void Depends::eval(
    std::vector<array>& outputs) {
  assert(inputs.size() > outputs.size());
  for (int i = 0; i < outputs.size(); i++) {
-    move_or_copy(inputs[i], outputs[i]);
+    outputs[i].copy_shared_buffer(inputs[i]);
  }
 }

@@ -98,7 +98,7 @@ void ExpandDims::eval(const std::vector<array>& inputs, array& out) {
  for (auto ax : axes_) {
    strides.insert(strides.begin() + ax, 1);
  }
-  move_or_copy(in, out, strides, in.flags(), in.data_size());
+  out.copy_shared_buffer(in, strides, in.flags(), in.data_size());
 }

 void NumberOfElements::eval(const std::vector<array>& inputs, array& out) {
@@ -210,7 +210,7 @@ void shared_buffer_reshape(
    auto max_dim = std::max_element(out.shape().begin(), out.shape().end());
    flags.col_contiguous = out.size() <= 1 || out.size() == *max_dim;
  }
-  move_or_copy(in, out, out_strides, flags, in.data_size());
+  out.copy_shared_buffer(in, out_strides, flags, in.data_size());
 }

 void Split::eval(
@@ -276,12 +276,12 @@ void Squeeze::eval(const std::vector<array>& inputs, array& out) {
      strides.push_back(in.strides(i));
    }
  }
-  move_or_copy(in, out, strides, in.flags(), in.data_size());
+  out.copy_shared_buffer(in, strides, in.flags(), in.data_size());
 }

 void StopGradient::eval(const std::vector<array>& inputs, array& out) {
  assert(inputs.size() == 1);
-  move_or_copy(inputs[0], out);
+  out.copy_shared_buffer(inputs[0]);
 }

 void Transpose::eval(const std::vector<array>& inputs, array& out) {
@@ -315,7 +315,7 @@ void Transpose::eval(const std::vector<array>& inputs, array& out) {
      b_stride *= out.shape(ri);
    }
  }
-  move_or_copy(in, out, out_strides, flags, in.data_size());
+  out.copy_shared_buffer(in, out_strides, flags, in.data_size());
 }

 } // namespace mlx::core
--- a/mlx/backend/common/compiled.cpp
+++ b/mlx/backend/common/compiled.cpp
@@ -161,8 +161,7 @@ void compiled_allocate_outputs(
    std::vector<array>& outputs,
    const std::vector<array>& inputs_,
    const std::unordered_set<uintptr_t>& constant_ids_,
-    bool contiguous,
-    bool move_buffers /* = false */) {
+    bool contiguous) {
  if (contiguous) {
    int o = 0;
    Strides strides;
@@ -178,11 +177,7 @@ void compiled_allocate_outputs(
      if (in.itemsize() == outputs[o].itemsize() && !is_scalar(in) &&
          in.is_donatable() &&
          constant_ids_.find(inputs_[i].id()) == constant_ids_.end()) {
-        if (move_buffers) {
-          outputs[o++].move_shared_buffer(in);
-        } else {
-          outputs[o++].copy_shared_buffer(in);
-        }
+        outputs[o++].copy_shared_buffer(in);
      }
      // Get representative input flags to properly set non-donated outputs
      if (strides.empty() && in.size() == outputs[0].size()) {
@@ -210,13 +205,8 @@ void compiled_allocate_outputs(
      if (in.flags().row_contiguous && in.size() == outputs[o].size() &&
          in.itemsize() == outputs[o].itemsize() && in.is_donatable() &&
          constant_ids_.find(inputs_[i].id()) == constant_ids_.end()) {
-        if (move_buffers) {
-          outputs[o].move_shared_buffer(
-              in, outputs[o].strides(), in.flags(), in.data_size());
-        } else {
-          outputs[o].copy_shared_buffer(
-              in, outputs[o].strides(), in.flags(), in.data_size());
-        }
+        outputs[o].copy_shared_buffer(
+            in, outputs[o].strides(), in.flags(), in.data_size());
        o++;
      }
    }
--- a/mlx/backend/common/compiled.h
+++ b/mlx/backend/common/compiled.h
@@ -62,7 +62,6 @@ void compiled_allocate_outputs(
    std::vector<array>& outputs,
    const std::vector<array>& inputs_,
    const std::unordered_set<uintptr_t>& constant_ids_,
-    bool contiguous,
-    bool move_buffers = false);
+    bool contiguous);

 } // namespace mlx::core
--- a/mlx/backend/common/copy.h
+++ b/mlx/backend/common/copy.h
@@ -22,4 +22,25 @@ enum class CopyType {
  GeneralGeneral
 };

+inline bool set_copy_output_data(const array& in, array& out, CopyType ctype) {
+  if (ctype == CopyType::Vector) {
+    // If the input is donateable, we are doing a vector copy and the types
+    // have the same size, then the input buffer can hold the output.
+    if (in.is_donatable() && in.itemsize() == out.itemsize()) {
+      out.copy_shared_buffer(in);
+      return true;
+    } else {
+      out.set_data(
+          allocator::malloc_or_wait(in.data_size() * out.itemsize()),
+          in.data_size(),
+          in.strides(),
+          in.flags());
+      return false;
+    }
+  } else {
+    out.set_data(allocator::malloc_or_wait(out.nbytes()));
+    return false;
+  }
+}
+
 } // namespace mlx::core
--- a/mlx/backend/common/load.cpp
+++ b/mlx/backend/common/load.cpp
@@ -3,7 +3,8 @@
 #include <algorithm>
 #include <utility>

-#include "mlx/backend/common/load.h"
+#include "mlx/primitives.h"
+#include "mlx/scheduler.h"

 namespace {

@@ -26,26 +27,31 @@ void swap_endianness(uint8_t* data_bytes, size_t N) {

 namespace mlx::core {

-void load(
-    array& out,
-    size_t offset,
-    const std::shared_ptr<io::Reader>& reader,
-    bool swap_endianness_) {
-  reader->read(out.data<char>(), out.nbytes(), offset);
-
-  if (swap_endianness_) {
-    switch (out.itemsize()) {
-      case 2:
-        swap_endianness<2>(out.data<uint8_t>(), out.data_size());
-        break;
-      case 4:
-        swap_endianness<4>(out.data<uint8_t>(), out.data_size());
-        break;
-      case 8:
-        swap_endianness<8>(out.data<uint8_t>(), out.data_size());
-        break;
+void Load::eval_cpu(const std::vector<array>& inputs, array& out) {
+  out.set_data(allocator::malloc_or_wait(out.nbytes()));
+  auto read_task = [out_ptr = out.data<char>(),
+                    size = out.size(),
+                    itemsize = out.itemsize(),
+                    offset = offset_,
+                    reader = reader_,
+                    swap_endianness_ = swap_endianness_]() mutable {
+    reader->read(out_ptr, size * itemsize, offset);
+    if (swap_endianness_) {
+      switch (itemsize) {
+        case 2:
+          swap_endianness<2>(reinterpret_cast<uint8_t*>(out_ptr), size);
+          break;
+        case 4:
+          swap_endianness<4>(reinterpret_cast<uint8_t*>(out_ptr), size);
+          break;
+        case 8:
+          swap_endianness<8>(reinterpret_cast<uint8_t*>(out_ptr), size);
+          break;
+      }
    }
-  }
+  };
+  auto fut = io::thread_pool().enqueue(std::move(read_task)).share();
+  scheduler::enqueue(stream(), [fut = std::move(fut)]() { fut.wait(); });
 }

 } // namespace mlx::core
--- a/mlx/backend/common/load.h
+++ b/mlx/backend/common/load.h
@@ -1,14 +0,0 @@
-// Copyright © 2024 Apple Inc.
-
-#include "mlx/array.h"
-#include "mlx/io/load.h"
-
-namespace mlx::core {
-
-void load(
-    array& out,
-    size_t offset,
-    const std::shared_ptr<io::Reader>& reader,
-    bool swap_endianess);
-
-} // namespace mlx::core
--- a/mlx/backend/common/slicing.cpp
+++ b/mlx/backend/common/slicing.cpp
@@ -36,7 +36,7 @@ void shared_buffer_slice(
  flags.col_contiguous = is_col_contiguous;
  flags.contiguous = (no_bsx_size == data_size);

-  move_or_copy(in, out, out_strides, flags, data_size, data_offset);
+  out.copy_shared_buffer(in, out_strides, flags, data_size, data_offset);
 }

 void slice(
--- a/mlx/backend/common/ternary.h
+++ b/mlx/backend/common/ternary.h
@@ -36,15 +36,10 @@ inline void set_ternary_op_output_data(
    const array& b,
    const array& c,
    array& out,
-    TernaryOpType topt,
-    bool donate_with_move = false) {
-  auto maybe_donate = [&out, donate_with_move](const array& x) {
+    TernaryOpType topt) {
+  auto maybe_donate = [&out](const array& x) {
    if (is_donatable(x, out)) {
-      if (donate_with_move) {
-        out.move_shared_buffer(x);
-      } else {
-        out.copy_shared_buffer(x);
-      }
+      out.copy_shared_buffer(x);
      return true;
    }
    return false;
--- a/mlx/backend/common/utils.cpp
+++ b/mlx/backend/common/utils.cpp
@@ -4,28 +4,6 @@

 namespace mlx::core {

-void move_or_copy(const array& in, array& out) {
-  if (in.is_donatable()) {
-    out.move_shared_buffer(in);
-  } else {
-    out.copy_shared_buffer(in);
-  }
-}
-
-void move_or_copy(
-    const array& in,
-    array& out,
-    const Strides& strides,
-    array::Flags flags,
-    size_t data_size,
-    size_t offset /* = 0 */) {
-  if (in.is_donatable()) {
-    out.move_shared_buffer(in, strides, flags, data_size, offset);
-  } else {
-    out.copy_shared_buffer(in, strides, flags, data_size, offset);
-  }
-}
-
 std::tuple<Shape, std::vector<Strides>> collapse_contiguous_dims(
    const Shape& shape,
    const std::vector<Strides>& strides,
--- a/mlx/backend/common/utils.h
+++ b/mlx/backend/common/utils.h
@@ -159,15 +159,6 @@ inline bool is_donatable(const array& in, const array& out) {
      in.buffer_size() <= out.nbytes() + donation_extra;
 }

-void move_or_copy(const array& in, array& out);
-void move_or_copy(
-    const array& in,
-    array& out,
-    const Strides& strides,
-    array::Flags flags,
-    size_t data_size,
-    size_t offset = 0);
-
 std::pair<bool, Strides> prepare_reshape(const array& in, const array& out);

 void shared_buffer_reshape(