Add mode parameter for quantization (#2499)

* add mode parameter for quantization * mxfp4 quantize/dequantize + start of optional biases * mxfp4 works * speedup * cpu mxfp4 * fix * fix test tol * fix * refactor * add quant mode enum
2025-12-16 01:49:05 +08:00 · 2025-08-28 06:45:26 -07:00
parent 7ef8a6f2d5
commit 70560b6bd5
28 changed files with 3635 additions and 757 deletions
--- a/mlx/export.cpp
+++ b/mlx/export.cpp
@@ -335,7 +335,7 @@ struct PrimitiveFactory {
      SERIALIZE_PRIMITIVE(Cholesky),
      SERIALIZE_PRIMITIVE(Eig),
      SERIALIZE_PRIMITIVE(Eigh),
-      SERIALIZE_PRIMITIVE(AffineQuantize),
+      SERIALIZE_PRIMITIVE(Quantize),
      SERIALIZE_PRIMITIVE(RMSNorm),
      SERIALIZE_PRIMITIVE(RMSNormVJP),
      SERIALIZE_PRIMITIVE(LayerNorm),