Fused Affine Quantize/Dequantize ops (#1282)

* Add fast affine dequantize * add full quantize kernel * fused kernel with scale/bias computation * fix docstring * fix no jit error * fix test * test fix * reduce fast api to only affine_quantize
2025-12-16 01:49:05 +08:00 · 2024-07-29 15:11:38 -07:00
parent aa1d6cadad
commit c52d1600f0
11 changed files with 655 additions and 400 deletions
--- a/python/src/fast.cpp
+++ b/python/src/fast.cpp
@@ -2,6 +2,7 @@

 #include <nanobind/nanobind.h>
 #include <nanobind/stl/optional.h>
+#include <nanobind/stl/tuple.h>
 #include <nanobind/stl/variant.h>

 #include "mlx/fast.h"
@@ -138,4 +139,47 @@ void init_fast(nb::module_& parent_module) {
        Returns:
            array: The output array.
      )pbdoc");
+
+  m.def(
+      "affine_quantize",
+      nb::overload_cast<
+          const array&,
+          const array&,
+          const array&,
+          int,
+          int,
+          StreamOrDevice>(&fast::affine_quantize),
+      "w"_a,
+      "scales"_a,
+      "biases"_a,
+      "group_size"_a = 64,
+      "bits"_a = 4,
+      nb::kw_only(),
+      "stream"_a = nb::none(),
+      nb::sig(
+          "def affine_quantize(w: array, /, scales: array, biases: array, group_size: int = 64, bits: int = 4, *, stream: Union[None, Stream, Device] = None) -> array"),
+      R"pbdoc(
+        Quantize the matrix ``w`` using the provided ``scales`` and
+        ``biases`` and the ``group_size`` and ``bits`` configuration.
+
+        Formally, given the notation in :func:`quantize`, we compute
+        :math:`w_i` from :math:`\hat{w_i}` and corresponding :math:`s` and
+        :math:`\beta` as follows
+
+        .. math::
+
+          w_i = s (\hat{w_i} + \beta)
+
+        Args:
+          w (array): Matrix to be quantize
+          scales (array): The scales to use per ``group_size`` elements of ``w``
+          biases (array): The biases to use per ``group_size`` elements of ``w``
+          group_size (int, optional): The size of the group in ``w`` that shares a
+            scale and bias. (default: ``64``)
+          bits (int, optional): The number of bits occupied by each element in
+            ``w``. (default: ``4``)
+
+        Returns:
+          array: The quantized version of ``w``
+      )pbdoc");
 }