Add sdpa with sinks (#2558)

* add sdpa with sinks * fix 2 pass * fix matrix sdpa * fix perf regression * add to cuda (#2580)
2025-12-16 01:49:05 +08:00 · 2025-09-10 14:53:00 -07:00
parent db5443e831
commit d6977f2a57
9 changed files with 349 additions and 114 deletions
--- a/python/src/fast.cpp
+++ b/python/src/fast.cpp
@@ -196,6 +196,7 @@ void init_fast(nb::module_& parent_module) {
         const mx::array& values,
         const float scale,
         const std::variant<std::monostate, std::string, mx::array>& mask,
+         const std::optional<mx::array>& sinks,
         mx::StreamOrDevice s) {
        bool has_mask = !std::holds_alternative<std::monostate>(mask);
        bool has_str_mask =
@@ -212,16 +213,16 @@ void init_fast(nb::module_& parent_module) {
              throw std::invalid_argument(msg.str());
            }
            return mx::fast::scaled_dot_product_attention(
-                queries, keys, values, scale, mask_str, {}, s);
+                queries, keys, values, scale, mask_str, {}, sinks, s);
          } else {
            auto mask_arr = std::get<mx::array>(mask);
            return mx::fast::scaled_dot_product_attention(
-                queries, keys, values, scale, "", {mask_arr}, s);
+                queries, keys, values, scale, "", {mask_arr}, sinks, s);
          }

        } else {
          return mx::fast::scaled_dot_product_attention(
-              queries, keys, values, scale, "", {}, s);
+              queries, keys, values, scale, "", {}, sinks, s);
        }
      },
      "q"_a,
@@ -230,9 +231,10 @@ void init_fast(nb::module_& parent_module) {
      nb::kw_only(),
      "scale"_a,
      "mask"_a = nb::none(),
+      "sinks"_a = nb::none(),
      "stream"_a = nb::none(),
      nb::sig(
-          "def scaled_dot_product_attention(q: array, k: array, v: array, *, scale: float,  mask: Union[None, str, array] = None, stream: Union[None, Stream, Device] = None) -> array"),
+          "def scaled_dot_product_attention(q: array, k: array, v: array, *, scale: float,  mask: Union[None, str, array] = None, sinks: Optional[array] = None, stream: Union[None, Stream, Device] = None) -> array"),
      R"pbdoc(
        A fast implementation of multi-head attention: ``O = softmax(Q @ K.T, dim=-1) @ V``.

@@ -262,14 +264,17 @@ void init_fast(nb::module_& parent_module) {
            q (array): Queries with shape ``[B, N_q, T_q, D]``.
            k (array): Keys with shape ``[B, N_kv, T_kv, D]``.
            v (array): Values with shape ``[B, N_kv, T_kv, D]``.
-            scale (float): Scale for queries (typically ``1.0 / sqrt(q.shape(-1)``)
-            mask (Union[None, str, array], optional): The mask to apply to the
+            scale (float): Scale for queries (typically ``1.0 / sqrt(q.shape(-1)``).
+            mask (str or array, optional): The mask to apply to the
               query-key scores. The mask can be an array or a string indicating
               the mask type. The only supported string type is ``"causal"``. If
               the mask is an array it can be a boolean or additive mask. The mask
               can have at most 4 dimensions and must be broadcast-compatible with
               the shape ``[B, N, T_q, T_kv]``. If an additive mask is given its
               type must promote to the promoted type of ``q``, ``k``, and ``v``.
+            sinks (array, optional): An optional array of attention sinks.
+               Default: ``None``.
+
        Returns:
            array: The output array.