Length masking for batch inputs (#1173)

* length masking * add mask to mlx_lm model interface * remove lengths * fix test: * comment + fix
2025-12-16 02:08:55 +08:00 · 2024-12-18 19:43:52 -08:00
parent db109184b7
commit d4ef909d4a
34 changed files with 191 additions and 72 deletions
--- a/llms/mlx_lm/models/qwen.py
+++ b/llms/mlx_lm/models/qwen.py
@@ -123,7 +123,8 @@ class QwenModel(nn.Module):
    def __call__(self, inputs, mask=None, cache=None):
        x = self.wte(inputs)

-        mask = create_attention_mask(x, cache)
+        if mask is None:
+            mask = create_attention_mask(x, cache)

        if cache is None:
            cache = [None] * len(self.h)