[CUDA] Switch to CUDA graphs (#2317)

* cuda graph prototype fix signal bug + start to add dependencies capture more capture more ops remaining ops fix reduce and rope deps add concurrent context try update, but not working cosistent topology order use node api use node api directly to reduce overhead fix bug use kernels in unary cache graph format fix synchronization format * comment
2025-12-16 01:49:05 +08:00 · 2025-07-02 15:59:13 -07:00
parent e76e9b87f0
commit ec0d5db67b
36 changed files with 1461 additions and 1212 deletions
--- a/mlx/backend/cuda/binary_two.cu
+++ b/mlx/backend/cuda/binary_two.cu
@@ -137,98 +137,101 @@ void binary_op_gpu_inplace(
  encoder.set_input_array(b);
  encoder.set_output_array(out_a);
  encoder.set_output_array(out_b);
-  encoder.launch_kernel([&](cudaStream_t stream) {
-    dispatch_all_types(a.dtype(), [&](auto in_type_tag) {
-      dispatch_all_types(out_a.dtype(), [&](auto out_type_tag) {
-        using CTYPE_IN = MLX_GET_TYPE(in_type_tag);
-        using CTYPE_OUT = MLX_GET_TYPE(out_type_tag);
-        if constexpr (cu::supports_binary_op<Op, CTYPE_IN, CTYPE_OUT>()) {
-          using InType = cuda_type_t<CTYPE_IN>;
-          using OutType = cuda_type_t<CTYPE_OUT>;
+  dispatch_all_types(a.dtype(), [&](auto in_type_tag) {
+    dispatch_all_types(out_a.dtype(), [&](auto out_type_tag) {
+      using CTYPE_IN = MLX_GET_TYPE(in_type_tag);
+      using CTYPE_OUT = MLX_GET_TYPE(out_type_tag);
+      if constexpr (cu::supports_binary_op<Op, CTYPE_IN, CTYPE_OUT>()) {
+        using InType = cuda_type_t<CTYPE_IN>;
+        using OutType = cuda_type_t<CTYPE_OUT>;

-          auto bopt = get_binary_op_type(a, b);
-          if (bopt == BinaryOpType::General) {
-            dispatch_bool(
-                a.data_size() > INT32_MAX || b.data_size() > INT32_MAX ||
-                    out_a.data_size() > INT32_MAX,
-                [&](auto large) {
-                  using IdxT = std::conditional_t<large(), int64_t, int32_t>;
-                  Shape shape;
-                  std::vector<Strides> strides;
-                  std::tie(shape, strides) =
-                      collapse_contiguous_dims(a, b, out_a);
-                  auto& a_strides = strides[0];
-                  auto& b_strides = strides[1];
-                  int ndim = shape.size();
-                  if (ndim <= 3) {
-                    dispatch_1_2_3(ndim, [&](auto dims_constant) {
-                      auto kernel = cu::binary_g_nd<
-                          Op,
-                          InType,
-                          OutType,
-                          IdxT,
-                          dims_constant()>;
-                      auto [num_blocks, block_dims] =
-                          get_launch_args(kernel, out_a, large());
-                      kernel<<<num_blocks, block_dims, 0, stream>>>(
-                          a.data<InType>(),
-                          b.data<InType>(),
-                          out_a.data<OutType>(),
-                          out_b.data<OutType>(),
-                          out_a.size(),
-                          const_param<dims_constant()>(shape),
-                          const_param<dims_constant()>(a_strides),
-                          const_param<dims_constant()>(b_strides));
-                    });
-                  } else {
-                    auto kernel = cu::binary_g<Op, InType, OutType, IdxT>;
+        auto bopt = get_binary_op_type(a, b);
+        if (bopt == BinaryOpType::General) {
+          dispatch_bool(
+              a.data_size() > INT32_MAX || b.data_size() > INT32_MAX ||
+                  out_a.data_size() > INT32_MAX,
+              [&](auto large) {
+                using IdxT = std::conditional_t<large(), int64_t, int32_t>;
+                Shape shape;
+                std::vector<Strides> strides;
+                std::tie(shape, strides) =
+                    collapse_contiguous_dims(a, b, out_a);
+                auto& a_strides = strides[0];
+                auto& b_strides = strides[1];
+                int ndim = shape.size();
+                if (ndim <= 3) {
+                  dispatch_1_2_3(ndim, [&](auto dims_constant) {
+                    auto kernel = cu::
+                        binary_g_nd<Op, InType, OutType, IdxT, dims_constant()>;
                    auto [num_blocks, block_dims] =
                        get_launch_args(kernel, out_a, large());
-                    kernel<<<num_blocks, block_dims, 0, stream>>>(
+                    encoder.add_kernel_node(
+                        kernel,
+                        num_blocks,
+                        block_dims,
                        a.data<InType>(),
                        b.data<InType>(),
                        out_a.data<OutType>(),
                        out_b.data<OutType>(),
                        out_a.size(),
-                        const_param(shape),
-                        const_param(a_strides),
-                        const_param(b_strides),
-                        ndim);
-                  }
-                });
-          } else {
-            dispatch_bool(out_a.data_size() > INT32_MAX, [&](auto large) {
-              using IdxT = std::conditional_t<large(), int64_t, uint32_t>;
-              auto kernel = cu::binary_ss<Op, InType, OutType, IdxT>;
-              if (bopt == BinaryOpType::ScalarVector) {
-                kernel = cu::binary_sv<Op, InType, OutType, IdxT>;
-              } else if (bopt == BinaryOpType::VectorScalar) {
-                kernel = cu::binary_vs<Op, InType, OutType, IdxT>;
-              } else if (bopt == BinaryOpType::VectorVector) {
-                kernel = cu::binary_vv<Op, InType, OutType, IdxT>;
-              }
-              auto [num_blocks, block_dims] = get_launch_args(
-                  kernel,
-                  out_a.data_size(),
-                  out_a.shape(),
-                  out_a.strides(),
-                  large());
-              kernel<<<num_blocks, block_dims, 0, stream>>>(
-                  a.data<InType>(),
-                  b.data<InType>(),
-                  out_a.data<OutType>(),
-                  out_b.data<OutType>(),
-                  out_a.data_size());
-            });
-          }
+                        const_param<dims_constant()>(shape),
+                        const_param<dims_constant()>(a_strides),
+                        const_param<dims_constant()>(b_strides));
+                  });
+                } else {
+                  auto kernel = cu::binary_g<Op, InType, OutType, IdxT>;
+                  auto [num_blocks, block_dims] =
+                      get_launch_args(kernel, out_a, large());
+                  encoder.add_kernel_node(
+                      kernel,
+                      num_blocks,
+                      block_dims,
+                      a.data<InType>(),
+                      b.data<InType>(),
+                      out_a.data<OutType>(),
+                      out_b.data<OutType>(),
+                      out_a.size(),
+                      const_param(shape),
+                      const_param(a_strides),
+                      const_param(b_strides),
+                      ndim);
+                }
+              });
        } else {
-          throw std::runtime_error(fmt::format(
-              "Can not do binary op {} on inputs of {} with result of {}.",
-              op,
-              dtype_to_string(a.dtype()),
-              dtype_to_string(out_a.dtype())));
+          dispatch_bool(out_a.data_size() > INT32_MAX, [&](auto large) {
+            using IdxT = std::conditional_t<large(), int64_t, uint32_t>;
+            auto kernel = cu::binary_ss<Op, InType, OutType, IdxT>;
+            if (bopt == BinaryOpType::ScalarVector) {
+              kernel = cu::binary_sv<Op, InType, OutType, IdxT>;
+            } else if (bopt == BinaryOpType::VectorScalar) {
+              kernel = cu::binary_vs<Op, InType, OutType, IdxT>;
+            } else if (bopt == BinaryOpType::VectorVector) {
+              kernel = cu::binary_vv<Op, InType, OutType, IdxT>;
+            }
+            auto [num_blocks, block_dims] = get_launch_args(
+                kernel,
+                out_a.data_size(),
+                out_a.shape(),
+                out_a.strides(),
+                large());
+            encoder.add_kernel_node(
+                kernel,
+                num_blocks,
+                block_dims,
+                a.data<InType>(),
+                b.data<InType>(),
+                out_a.data<OutType>(),
+                out_b.data<OutType>(),
+                out_a.data_size());
+          });
        }
-      });
+      } else {
+        throw std::runtime_error(fmt::format(
+            "Can not do binary op {} on inputs of {} with result of {}.",
+            op,
+            dtype_to_string(a.dtype()),
+            dtype_to_string(out_a.dtype())));
+      }
    });
  });
 }