CUDA backend: binary ops

2025-12-16 01:49:05 +08:00 · 2025-04-13 23:51:11 +00:00
parent df6fa3914b
commit 194212f65f
6 changed files with 565 additions and 19 deletions
--- a/mlx/backend/cuda/CMakeLists.txt
+++ b/mlx/backend/cuda/CMakeLists.txt
@@ -6,6 +6,7 @@
 target_sources(
  mlx
  PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/allocator.cpp
+          ${CMAKE_CURRENT_SOURCE_DIR}/binary.cu
          ${CMAKE_CURRENT_SOURCE_DIR}/copy.cpp
          ${CMAKE_CURRENT_SOURCE_DIR}/device.cpp
          ${CMAKE_CURRENT_SOURCE_DIR}/eval.cpp
--- a/mlx/backend/cuda/binary.cu
+++ b/mlx/backend/cuda/binary.cu
@@ -0,0 +1,212 @@
+// Copyright © 2025 Apple Inc.
+
+#include "mlx/backend/common/binary.h"
+#include "mlx/backend/cuda/device.h"
+#include "mlx/backend/cuda/iterators/general_iterator.cuh"
+#include "mlx/backend/cuda/iterators/repeat_iterator.cuh"
+#include "mlx/backend/cuda/kernel_utils.cuh"
+#include "mlx/backend/cuda/kernels/binary_ops.cuh"
+#include "mlx/backend/cuda/kernels/cucomplex_math.cuh"
+#include "mlx/dtype_utils.h"
+#include "mlx/primitives.h"
+
+#include <nvtx3/nvtx3.hpp>
+#include <thrust/copy.h>
+#include <thrust/device_ptr.h>
+#include <thrust/transform.h>
+
+namespace mlx::core {
+
+namespace cu {
+
+template <typename Op, typename In, typename Out>
+constexpr bool supports_binary_op() {
+  if (std::is_same_v<Op, Add> || std::is_same_v<Op, Divide> ||
+      std::is_same_v<Op, Maximum> || std::is_same_v<Op, Minimum> ||
+      std::is_same_v<Op, Multiply> || std::is_same_v<Op, Subtract> ||
+      std::is_same_v<Op, Power> || std::is_same_v<Op, Remainder>) {
+    return std::is_same_v<In, Out>;
+  }
+  if (std::is_same_v<Op, Equal> || std::is_same_v<Op, Greater> ||
+      std::is_same_v<Op, GreaterEqual> || std::is_same_v<Op, Less> ||
+      std::is_same_v<Op, LessEqual> || std::is_same_v<Op, NotEqual>) {
+    return std::is_same_v<Out, bool>;
+  }
+  if (std::is_same_v<Op, LogicalAnd> || std::is_same_v<Op, LogicalOr>) {
+    return std::is_same_v<Out, bool> && std::is_same_v<In, bool>;
+  }
+  if (std::is_same_v<Op, NaNEqual>) {
+    return std::is_same_v<Out, bool> &&
+        (is_floating_v<In> || std::is_same_v<In, complex64_t>);
+  }
+  if (std::is_same_v<Op, LogAddExp> || std::is_same_v<Op, ArcTan2>) {
+    return std::is_same_v<In, Out> && is_floating_v<In>;
+  }
+  if (std::is_same_v<Op, BitwiseAnd> || std::is_same_v<Op, BitwiseOr> ||
+      std::is_same_v<Op, BitwiseXor>) {
+    return std::is_same_v<In, Out> && std::is_integral_v<In>;
+  }
+  if (std::is_same_v<Op, LeftShift> || std::is_same_v<Op, RightShift>) {
+    return std::is_same_v<In, Out> && std::is_integral_v<In> &&
+        !std::is_same_v<In, bool>;
+  }
+  return false;
+}
+
+} // namespace cu
+
+template <typename Op>
+void binary_op_gpu_inplace(
+    const std::vector<array>& inputs,
+    std::vector<array>& outputs,
+    std::string_view op,
+    const Stream& s) {
+  auto& a = inputs[0];
+  auto& b = inputs[1];
+  auto& out = outputs[0];
+  if (out.size() == 0) {
+    return;
+  }
+
+  auto& encoder = cu::get_command_encoder(s);
+  encoder.set_input_array(a);
+  encoder.set_input_array(b);
+  encoder.set_output_array(out);
+  encoder.launch_kernel([&](cudaStream_t stream) {
+    MLX_SWITCH_ALL_TYPES(a.dtype(), CTYPE_IN, {
+      MLX_SWITCH_ALL_TYPES(out.dtype(), CTYPE_OUT, {
+        if constexpr (cu::supports_binary_op<Op, CTYPE_IN, CTYPE_OUT>()) {
+          using InType = cuda_type_t<CTYPE_IN>;
+          using OutType = cuda_type_t<CTYPE_OUT>;
+          auto policy = cu::thrust_policy(stream);
+          auto a_ptr = thrust::device_pointer_cast(a.data<InType>());
+          auto b_ptr = thrust::device_pointer_cast(b.data<InType>());
+          auto out_ptr = thrust::device_pointer_cast(out.data<OutType>());
+
+          auto bopt = get_binary_op_type(a, b);
+          if (bopt == BinaryOpType::ScalarScalar) {
+            auto a_begin = cu::repeat_iterator(a_ptr);
+            auto a_end = a_begin + out.data_size();
+            auto b_begin = cu::repeat_iterator(b_ptr);
+            thrust::transform(policy, a_begin, a_end, b_begin, out_ptr, Op());
+          } else if (bopt == BinaryOpType::ScalarVector) {
+            auto a_begin = cu::repeat_iterator(a_ptr);
+            auto a_end = a_begin + out.data_size();
+            auto b_begin = b_ptr;
+            thrust::transform(policy, a_begin, a_end, b_begin, out_ptr, Op());
+          } else if (bopt == BinaryOpType::VectorScalar) {
+            auto a_begin = a_ptr;
+            auto a_end = a_begin + out.data_size();
+            auto b_begin = cu::repeat_iterator(b_ptr);
+            thrust::transform(policy, a_begin, a_end, b_begin, out_ptr, Op());
+          } else if (bopt == BinaryOpType::VectorVector) {
+            auto a_begin = a_ptr;
+            auto a_end = a_begin + out.data_size();
+            auto b_begin = b_ptr;
+            thrust::transform(policy, a_begin, a_end, b_begin, out_ptr, Op());
+          } else {
+            auto [shape, strides] = collapse_contiguous_dims(a, b, out);
+            auto [a_begin, a_end] = cu::make_general_iterators<int64_t>(
+                a_ptr, out.data_size(), shape, strides[0]);
+            auto [b_begin, b_end] = cu::make_general_iterators<int64_t>(
+                b_ptr, out.data_size(), shape, strides[1]);
+            thrust::transform(policy, a_begin, a_end, b_begin, out_ptr, Op());
+          }
+        } else {
+          throw std::runtime_error(fmt::format(
+              "Can not do binary op {} on inputs of {} with result of {}.",
+              op,
+              dtype_to_string(a.dtype()),
+              dtype_to_string(out.dtype())));
+        }
+      });
+    });
+  });
+}
+
+template <typename Op>
+void binary_op_gpu(
+    const std::vector<array>& inputs,
+    std::vector<array>& outputs,
+    std::string_view op,
+    const Stream& s) {
+  auto& a = inputs[0];
+  auto& b = inputs[1];
+  auto bopt = get_binary_op_type(a, b);
+  set_binary_op_output_data(a, b, outputs[0], bopt);
+  set_binary_op_output_data(a, b, outputs[1], bopt);
+  binary_op_gpu_inplace<Op>(inputs, outputs, op, s);
+}
+
+template <typename Op>
+void binary_op_gpu(
+    const std::vector<array>& inputs,
+    array& out,
+    std::string_view op,
+    const Stream& s) {
+  auto& a = inputs[0];
+  auto& b = inputs[1];
+  auto bopt = get_binary_op_type(a, b);
+  set_binary_op_output_data(a, b, out, bopt);
+  std::vector<array> outputs{out};
+  binary_op_gpu_inplace<Op>(inputs, outputs, op, s);
+}
+
+#define BINARY_GPU(func)                                                 \
+  void func::eval_gpu(const std::vector<array>& inputs, array& out) {    \
+    nvtx3::scoped_range r(#func "::eval_gpu");                           \
+    auto& s = out.primitive().stream();                                  \
+    binary_op_gpu<cu::func>(inputs, out, get_primitive_string(this), s); \
+  }
+
+#define BINARY_GPU_MULTI(func)                                               \
+  void func::eval_gpu(                                                       \
+      const std::vector<array>& inputs, std::vector<array>& outputs) {       \
+    nvtx3::scoped_range r(#func "::eval_gpu");                               \
+    auto& s = outputs[0].primitive().stream();                               \
+    binary_op_gpu<cu::func>(inputs, outputs, get_primitive_string(this), s); \
+  }
+
+BINARY_GPU(Add)
+BINARY_GPU(ArcTan2)
+BINARY_GPU(Divide)
+BINARY_GPU(Remainder)
+BINARY_GPU(Equal)
+BINARY_GPU(Greater)
+BINARY_GPU(GreaterEqual)
+BINARY_GPU(Less)
+BINARY_GPU(LessEqual)
+BINARY_GPU(LogicalAnd)
+BINARY_GPU(LogicalOr)
+BINARY_GPU(LogAddExp)
+BINARY_GPU(Maximum)
+BINARY_GPU(Minimum)
+BINARY_GPU(Multiply)
+BINARY_GPU(NotEqual)
+BINARY_GPU(Power)
+BINARY_GPU(Subtract)
+
+void BitwiseBinary::eval_gpu(const std::vector<array>& inputs, array& out) {
+  nvtx3::scoped_range r("BitwiseBinary::eval_gpu");
+  auto& s = out.primitive().stream();
+  auto op = get_primitive_string(this);
+  switch (op_) {
+    case BitwiseBinary::And:
+      binary_op_gpu<cu::BitwiseAnd>(inputs, out, op, s);
+      break;
+    case BitwiseBinary::Or:
+      binary_op_gpu<cu::BitwiseOr>(inputs, out, op, s);
+      break;
+    case BitwiseBinary::Xor:
+      binary_op_gpu<cu::BitwiseXor>(inputs, out, op, s);
+      break;
+    case BitwiseBinary::LeftShift:
+      binary_op_gpu<cu::LeftShift>(inputs, out, op, s);
+      break;
+    case BitwiseBinary::RightShift:
+      binary_op_gpu<cu::RightShift>(inputs, out, op, s);
+      break;
+  }
+}
+
+} // namespace mlx::core
--- a/mlx/backend/cuda/iterators/repeat_iterator.cuh
+++ b/mlx/backend/cuda/iterators/repeat_iterator.cuh
@@ -0,0 +1,31 @@
+// Copyright © 2025 Apple Inc.
+
+#pragma once
+
+#include <thrust/iterator/iterator_adaptor.h>
+
+namespace mlx::core::cu {
+
+// Always return the value of initial iterator after advancements.
+template <typename Iterator>
+class repeat_iterator
+    : public thrust::iterator_adaptor<repeat_iterator<Iterator>, Iterator> {
+ public:
+  using super_t = thrust::iterator_adaptor<repeat_iterator<Iterator>, Iterator>;
+  using reference = typename super_t::reference;
+  using difference_type = typename super_t::difference_type;
+
+  __host__ __device__ repeat_iterator(Iterator it) : super_t(it), it_(it) {}
+
+ private:
+  friend class thrust::iterator_core_access;
+
+  // The dereference is device-only to avoid accidental running in host.
+  __device__ typename super_t::reference dereference() const {
+    return *it_;
+  }
+
+  Iterator it_;
+};
+
+} // namespace mlx::core::cu
--- a/mlx/backend/cuda/kernels/binary_ops.cuh
+++ b/mlx/backend/cuda/kernels/binary_ops.cuh
@@ -0,0 +1,275 @@
+// Copyright © 2025 Apple Inc.
+
+#include "mlx/backend/cuda/kernels/fp16_math.cuh"
+
+namespace mlx::core::cu {
+
+struct Add {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x + y;
+  }
+};
+
+struct FloorDivide {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    if constexpr (cuda::std::is_integral_v<T>) {
+      return x / y;
+    } else {
+      return trunc(x / y);
+    }
+  }
+};
+
+struct Divide {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x / y;
+  }
+};
+
+struct Remainder {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    if constexpr (cuda::std::is_integral_v<T>) {
+      if constexpr (cuda::std::is_signed_v<T>) {
+        auto r = x % y;
+        if (r != 0 && (r < 0 != y < 0)) {
+          r += y;
+        }
+        return r;
+      } else {
+        return x % y;
+      }
+    } else if constexpr (cuda::std::is_same_v<T, cuComplex>) {
+      return x % y;
+    } else {
+      T r = fmod(x, y);
+      if (r != 0 && (r < 0 != y < 0)) {
+        r = r + y;
+      }
+      return r;
+    }
+  }
+};
+
+struct Equal {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    return x == y;
+  }
+};
+
+struct NaNEqual {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    if constexpr (std::is_same_v<T, cuComplex>) {
+      return x == y ||
+          (isnan(cuCrealf(x)) && isnan(cuCrealf(y)) && isnan(cuCimagf(x)) &&
+           isnan(cuCimagf(y))) ||
+          (cuCrealf(x) == cuCrealf(y) && isnan(cuCimagf(x)) &&
+           isnan(cuCimagf(y))) ||
+          (isnan(cuCrealf(x)) && isnan(cuCrealf(y)) &&
+           cuCimagf(x) == cuCimagf(y));
+    } else {
+      return x == y || (isnan(x) && isnan(y));
+    }
+  }
+};
+
+struct Greater {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    return x > y;
+  }
+};
+
+struct GreaterEqual {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    return x >= y;
+  }
+};
+
+struct Less {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    return x < y;
+  }
+};
+
+struct LessEqual {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    return x <= y;
+  }
+};
+
+struct LogAddExp {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    if (isnan(x) || isnan(y)) {
+      return cuda::std::numeric_limits<T>::quiet_NaN();
+    }
+    T maxval = max(x, y);
+    T minval = min(x, y);
+    return (minval == -cuda::std::numeric_limits<T>::infinity() ||
+            maxval == cuda::std::numeric_limits<T>::infinity())
+        ? maxval
+        : T(float(maxval) + log1p(expf(minval - maxval)));
+  };
+};
+
+struct Maximum {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    if constexpr (cuda::std::is_integral_v<T>) {
+      return max(x, y);
+    } else if constexpr (cuda::std::is_same_v<T, cuComplex>) {
+      if (isnan(cuCrealf(x)) || isnan(cuCimagf(x))) {
+        return x;
+      }
+      return x > y ? x : y;
+    } else {
+      if (isnan(x)) {
+        return x;
+      }
+      return x > y ? x : y;
+    }
+  }
+};
+
+struct Minimum {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    if constexpr (cuda::std::is_integral_v<T>) {
+      return min(x, y);
+    } else if constexpr (cuda::std::is_same_v<T, cuComplex>) {
+      if (isnan(cuCrealf(x)) || isnan(cuCimagf(x))) {
+        return x;
+      }
+      return x < y ? x : y;
+    } else {
+      if (isnan(x)) {
+        return x;
+      }
+      return x < y ? x : y;
+    }
+  }
+};
+
+struct Multiply {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x * y;
+  }
+};
+
+struct NotEqual {
+  template <typename T>
+  __device__ bool operator()(T x, T y) {
+    if constexpr (std::is_same_v<T, cuComplex>) {
+      return cuCrealf(x) != cuCrealf(y) || cuCimagf(x) != cuCimagf(y);
+    } else {
+      return x != y;
+    }
+  }
+};
+
+struct Power {
+  template <typename T>
+  __device__ T operator()(T base, T exp) {
+    if constexpr (cuda::std::is_integral_v<T>) {
+      T res = 1;
+      while (exp) {
+        if (exp & 1) {
+          res *= base;
+        }
+        exp >>= 1;
+        base *= base;
+      }
+      return res;
+    } else if constexpr (cuda::std::is_same_v<T, cuComplex>) {
+      auto x_theta = atan2f(base.y, base.x);
+      auto x_ln_r = 0.5 * logf(base.x * base.x + base.y * base.y);
+      auto mag = expf(exp.x * x_ln_r - exp.y * x_theta);
+      auto phase = exp.y * x_ln_r + exp.x * x_theta;
+      return make_cuFloatComplex(mag * cosf(phase), mag * sinf(phase));
+    } else {
+      return powf(base, exp);
+    }
+  }
+};
+
+struct Subtract {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x - y;
+  }
+};
+
+struct LogicalAnd {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x && y;
+  };
+};
+
+struct LogicalOr {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x || y;
+  };
+};
+
+struct BitwiseAnd {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x & y;
+  };
+};
+
+struct BitwiseOr {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x | y;
+  };
+};
+
+struct BitwiseXor {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x ^ y;
+  };
+};
+
+struct LeftShift {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x << y;
+  };
+};
+
+struct RightShift {
+  template <typename T>
+  __device__ T operator()(T x, T y) {
+    return x >> y;
+  };
+};
+
+struct ArcTan2 {
+  template <typename T>
+  __device__ T operator()(T y, T x) {
+    return atan2f(y, x);
+  }
+};
+
+struct DivMod {
+  template <typename T>
+  __device__ cuda::std::array<T, 2> operator()(T x, T y) {
+    return {FloorDivide{}(x, y), Remainder{}(x, y)};
+  };
+};
+
+} // namespace mlx::core::cu
--- a/mlx/backend/cuda/kernels/fp16_math.cuh
+++ b/mlx/backend/cuda/kernels/fp16_math.cuh
@@ -81,6 +81,52 @@ MLX_DEFINE_UNARY_OP_FALLBCK(tanh)
 #undef MLX_DEFINE_UNARY_OP
 #undef MLX_DEFINE_UNARY_OP_FALLBCK

+///////////////////////////////////////////////////////////////////////////////
+// Binary ops for half types.
+///////////////////////////////////////////////////////////////////////////////
+
+#if CUDART_VERSION < 12000 && __CUDA_ARCH__ < 800
+#define MLX_DEFINE_BINARY_OP(NAME, HALF_OP)          \
+  template <typename T>                              \
+  __forceinline__ __device__ auto NAME(T x, T y) {   \
+    if constexpr (cuda::std::is_same_v<T, __half>) { \
+      return HALF_OP(x, y);                          \
+    } else {                                         \
+      return ::NAME(x, y);                           \
+    }                                                \
+  }
+#else
+#define MLX_DEFINE_BINARY_OP(NAME, HALF_OP)                        \
+  template <typename T>                                            \
+  __forceinline__ __device__ auto NAME(T x, T y) {                 \
+    if constexpr (cuda::std::is_same_v<T, __half>) {               \
+      return HALF_OP(x, y);                                        \
+    } else if constexpr (cuda::std::is_same_v<T, __nv_bfloat16>) { \
+      return HALF_OP(x, y);                                        \
+    } else {                                                       \
+      return ::NAME(x, y);                                         \
+    }                                                              \
+  }
+#endif
+
+MLX_DEFINE_BINARY_OP(max, __hmax)
+MLX_DEFINE_BINARY_OP(min, __hmin)
+
+#undef MLX_DEFINE_BINARY_OP
+
+template <typename T>
+__forceinline__ __device__ T fmod(T x, T y) {
+  if constexpr (cuda::std::is_same_v<T, __half>) {
+    return __float2half(::fmod(__half2float(x), __half2float(y)));
+#if CUDART_VERSION >= 12000 || __CUDA_ARCH__ >= 800
+  } else if constexpr (cuda::std::is_same_v<T, __nv_bfloat16>) {
+    return __float2bfloat16(::fmod(__bfloat162float(x), __bfloat162float(y)));
+#endif
+  } else {
+    return ::fmod(x, y);
+  }
+}
+
 ///////////////////////////////////////////////////////////////////////////////
 // Additional C++ operator overrides between half types and native types.
 ///////////////////////////////////////////////////////////////////////////////
--- a/mlx/backend/cuda/primitives.cu
+++ b/mlx/backend/cuda/primitives.cu
@@ -54,45 +54,27 @@ void Arange::eval_gpu(const std::vector<array>& inputs, array& out) {
    throw std::runtime_error(#func " has no CUDA implementation.");   \
  }

-NO_GPU(Add)
 NO_GPU(AddMM)
-NO_GPU(ArcTan2)
 NO_GPU(ArgPartition)
 NO_GPU(ArgReduce)
 NO_GPU(ArgSort)
-NO_GPU(BitwiseBinary)
 NO_GPU(BlockMaskedMM)
 NO_GPU_MULTI(Compiled)
 NO_GPU(Convolution)
-NO_GPU(Divide)
 NO_GPU_MULTI(DivMod)
 NO_GPU(DynamicSlice)
 NO_GPU(DynamicSliceUpdate)
-NO_GPU(Remainder)
-NO_GPU(Equal)
 NO_GPU(FFT)
 NO_GPU(Gather)
 NO_GPU(GatherAxis)
 NO_GPU(GatherMM)
 NO_GPU(GatherQMM)
-NO_GPU(Greater)
-NO_GPU(GreaterEqual)
 NO_GPU(Hadamard)
-NO_GPU(Less)
-NO_GPU(LessEqual)
 NO_GPU(Load)
-NO_GPU(LogicalAnd)
-NO_GPU(LogicalOr)
-NO_GPU(LogAddExp)
 NO_GPU(LogSumExp)
 NO_GPU_MULTI(LUF)
 NO_GPU(Matmul)
-NO_GPU(Maximum)
-NO_GPU(Minimum)
-NO_GPU(Multiply)
-NO_GPU(NotEqual)
 NO_GPU(Partition)
-NO_GPU(Power)
 NO_GPU_MULTI(QRF)
 NO_GPU(QuantizedMatmul)
 NO_GPU(RandomBits)
@@ -104,7 +86,6 @@ NO_GPU(Select)
 NO_GPU(SliceUpdate)
 NO_GPU(Softmax)
 NO_GPU(Sort)
-NO_GPU(Subtract)
 NO_GPU_MULTI(SVD)
 NO_GPU(Inverse)
 NO_GPU(Cholesky)