Fix attention vizualizer (#40285)

molbap · web-flow · commit 04b751f07d61 · 2025-08-21T13:13:35.000Z
* make visualizer rely on create causal mask

* format

* fixup

* fixup

* read token

* read token, duh

* what is up with that token

* small tests?

* adjust

* try with flush

* normalize for ANSI

* buffer shenanigans
diff --git a/src/transformers/utils/attention_visualizer.py b/src/transformers/utils/attention_visualizer.py
@@ -16,6 +16,7 @@
 import requests
 from PIL import Image
 
+from ..masking_utils import create_causal_mask
 from ..models.auto.auto_factory import _get_model_class
 from ..models.auto.configuration_auto import AutoConfig
 from ..models.auto.modeling_auto import MODEL_FOR_PRETRAINING_MAPPING, MODEL_MAPPING
@@ -207,13 +208,23 @@ def visualize_attention_mask(self, input_sentence: str, suffix=""):
 
         model.config._attn_implementation = "eager"
         model.train()
-        attention_mask = ~model._update_causal_mask(
+
+        batch_size, seq_length = attention_mask.shape
+        input_embeds = torch.zeros((batch_size, seq_length, model.config.hidden_size), dtype=self.model.dtype)
+        cache_position = torch.arange(seq_length)
+
+        causal_mask = create_causal_mask(
+            config=model.config,
+            input_embeds=input_embeds,
             attention_mask=attention_mask,
-            input_tensor=attention_mask.to(self.model.dtype),
-            cache_position=torch.arange(attention_mask.shape[1]),
+            cache_position=cache_position,
             past_key_values=None,
-            **kwargs,
-        ).bool()
+        )
+
+        if causal_mask is not None:
+            attention_mask = ~causal_mask.bool()
+        else:
+            attention_mask = attention_mask.unsqueeze(1).unsqueeze(1).expand(batch_size, 1, seq_length, seq_length)
         top_bottom_border = "##" * (
             len(f"Attention visualization for {self.config.model_type} | {self.mapped_cls}") + 4
         )  # Box width adjusted to text length
@@ -225,7 +236,7 @@ def visualize_attention_mask(self, input_sentence: str, suffix=""):
                 len(top_bottom_border)
             )
             + "    "
-            + side_border
+            + side_border,
         )
         print(f"{top_bottom_border}")
         f_string = generate_attention_matrix_from_mask(
diff --git a/tests/utils/test_attention_visualizer.py b/tests/utils/test_attention_visualizer.py
@@ -0,0 +1,127 @@
+# Copyright 2025 The HuggingFace Inc. team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import builtins
+import io
+import re
+import unittest
+
+from transformers.testing_utils import require_read_token, require_torch
+from transformers.utils.attention_visualizer import AttentionMaskVisualizer
+
+
+ANSI_RE = re.compile(r"\x1b\[[0-9;]*m")
+
+
+def _normalize(s: str) -> str:
+    # drop ANSI (colors may be disabled on CI), normalize line endings,
+    # and strip trailing spaces without touching alignment inside lines
+    s = ANSI_RE.sub("", s)
+    s = s.replace("\r\n", "\n").replace("\r", "\n")
+    return "\n".join(line.rstrip() for line in s.split("\n")).strip()
+
+
+@require_torch
+class AttentionMaskVisualizerTester(unittest.TestCase):
+    """Test suite for AttentionMaskVisualizer"""
+
+    @require_read_token
+    def test_paligemma_multimodal_visualization(self):
+        """Test AttentionMaskVisualizer with PaliGemma multimodal model"""
+        model_name = "hf-internal-testing/namespace_google_repo_name_paligemma-3b-pt-224"
+        input_text = "<img> What is in this image?"
+
+        buf = io.StringIO()
+        orig_print = builtins.print
+
+        def _print(*args, **kwargs):
+            kwargs.setdefault("file", buf)
+            orig_print(*args, **kwargs)
+
+        try:
+            builtins.print = _print
+            visualizer = AttentionMaskVisualizer(model_name)
+            visualizer(input_text)
+        finally:
+            builtins.print = orig_print
+        output = buf.getvalue()
+
+        expected_output = """
+##########################################################################################################################################################################################################################################
+##                                                      Attention visualization for \033[1mpaligemma:hf-internal-testing/namespace_google_repo_name_paligemma-3b-pt-224\033[0m PaliGemmaModel                                                         ##
+##########################################################################################################################################################################################################################################
+ \033[92m■\033[0m: i == j (diagonal)   \033[93m■\033[0m: token_type_ids
+              Attention Matrix  
+
+
+\033[93m'<image>'\033[0m:  0 \033[93m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+\033[93m'<image>'\033[0m:  1 \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+\033[93m'<image>'\033[0m:  2 \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+\033[93m'<image>'\033[0m:  3 \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+\033[93m'<image>'\033[0m:  4 \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m \033[93m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'<bos>'  :  5 ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁What'  :  6 ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁is'    :  7 ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁in'    :  8 ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁this'  :  9 ■ ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚    |    
+'▁image' : 10 ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚    |    
+'?'      : 11 ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚    |    
+'\\n'     : 12 ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚    |    
+'<eos>'  : 13 ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m    |    
+##########################################################################################################################################################################################################################################
+"""  # noqa
+
+        self.assertEqual(_normalize(output), _normalize(expected_output))
+
+    @require_read_token
+    def test_llama_text_only_visualization(self):
+        """Test AttentionMaskVisualizer with Llama text-only model"""
+        model_name = "hf-internal-testing/namespace_meta-llama_repo_name_Llama-2-7b-hf"
+        input_text = "Plants create energy through a process known as"
+
+        buf = io.StringIO()
+        orig_print = builtins.print
+
+        def _print(*args, **kwargs):
+            kwargs.setdefault("file", buf)
+            orig_print(*args, **kwargs)
+
+        try:
+            builtins.print = _print
+            visualizer = AttentionMaskVisualizer(model_name)
+            visualizer(input_text)
+        finally:
+            builtins.print = orig_print
+        output = buf.getvalue()
+
+        expected_output = """
+##########################################################################################################################################################################################################
+##                                           Attention visualization for \033[1mllama:hf-internal-testing/namespace_meta-llama_repo_name_Llama-2-7b-hf\033[0m LlamaModel                                              ##
+##########################################################################################################################################################################################################
+ \033[92m■\033[0m: i == j (diagonal)   \033[93m■\033[0m: token_type_ids
+               Attention Matrix
+
+'▁Pl'     :  0 \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'ants'    :  1 ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁create' :  2 ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁energy' :  3 ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚ ⬚    |    
+'▁through':  4 ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚ ⬚    |    
+'▁a'      :  5 ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚ ⬚    |    
+'▁process':  6 ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚ ⬚    |    
+'▁known'  :  7 ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m ⬚    |    
+'▁as'     :  8 ■ ■ ■ ■ ■ ■ ■ ■ \033[92m■\033[0m    |    
+##########################################################################################################################################################################################################
+"""  # noqa
+
+        self.assertEqual(_normalize(output), _normalize(expected_output))