neuralmagic
diff --git a/‎.github/workflows/multi-gpu-e2e.yml‎
Lines changed: 5 additions & 5 deletions b/‎.github/workflows/multi-gpu-e2e.yml‎
Lines changed: 5 additions & 5 deletions
diff --git a/‎.github/workflows/tests.yml‎
Lines changed: 2 additions & 2 deletions b/‎.github/workflows/tests.yml‎
Lines changed: 2 additions & 2 deletions
diff --git a/‎examples/llama4/scout-lora.yaml‎
Lines changed: 75 additions & 0 deletions b/‎examples/llama4/scout-lora.yaml‎
Lines changed: 75 additions & 0 deletions
diff --git a/‎requirements.txt‎
Lines changed: 3 additions & 2 deletions b/‎requirements.txt‎
Lines changed: 3 additions & 2 deletions
diff --git a/‎src/axolotl/core/trainers/base.py‎
Lines changed: 13 additions & 0 deletions b/‎src/axolotl/core/trainers/base.py‎
Lines changed: 13 additions & 0 deletions
diff --git a/‎src/axolotl/integrations/liger/__init__.py‎
Lines changed: 12 additions & 0 deletions b/‎src/axolotl/integrations/liger/__init__.py‎
Lines changed: 12 additions & 0 deletions
diff --git a/‎src/axolotl/integrations/liger/models/llama4.py‎
Lines changed: 171 additions & 0 deletions b/‎src/axolotl/integrations/liger/models/llama4.py‎
Lines changed: 171 additions & 0 deletions
diff --git a/‎src/axolotl/monkeypatch/attention/flex_attn.py‎
Lines changed: 0 additions & 1 deletion b/‎src/axolotl/monkeypatch/attention/flex_attn.py‎
Lines changed: 0 additions & 1 deletion
diff --git a/‎src/axolotl/monkeypatch/multipack.py‎
Lines changed: 1 addition & 0 deletions b/‎src/axolotl/monkeypatch/multipack.py‎
Lines changed: 1 addition & 0 deletions
@@ -27,21 +27,21 @@ jobs:
           - cuda: 124
             cuda_version: 12.4.1
             python_version: "3.11"
-            pytorch: 2.4.1
-            axolotl_extras:  # no vllm support for 2.4.1
+            pytorch: 2.6.0
+            axolotl_extras: vllm
             num_gpus: 2
             nightly_build: "true"
           - cuda: 124
             cuda_version: 12.4.1
             python_version: "3.11"
-            pytorch: 2.5.1
-            axolotl_extras: vllm
+            pytorch: 2.4.1
+            axolotl_extras:  # no vllm support for 2.4.1
             num_gpus: 2
             nightly_build: "true"
           - cuda: 124
             cuda_version: 12.4.1
             python_version: "3.11"
-            pytorch: 2.6.0
+            pytorch: 2.5.1
             axolotl_extras: vllm
             num_gpus: 2
             nightly_build: "true"
 
@@ -211,7 +211,7 @@ jobs:
           - cuda: 124
             cuda_version: 12.4.1
             python_version: "3.11"
-            pytorch: 2.5.1
+            pytorch: 2.6.0
             num_gpus: 1
             axolotl_extras: vllm
     steps:
@@ -258,7 +258,7 @@ jobs:
           - cuda: 124
             cuda_version: 12.4.1
             python_version: "3.11"
-            pytorch: 2.6.0
+            pytorch: 2.5.1
             num_gpus: 1
             axolotl_extras: vllm
     steps:
 
@@ -0,0 +1,75 @@
+base_model: meta-llama/Llama-4-Scout-17B-16E
+model_type: Llama4ForConditionalGeneration
+  # Automatically upload checkpoint and final model to HF
+  # hub_model_id: username/custom_model_name
+
+strict: false
+
+  # torch_compile: true
+
+adapter: lora
+lora_r: 32
+lora_alpha: 64
+lora_target_modules:
+  - self_attn.q_proj
+  - self_attn.k_proj
+  - self_attn.v_proj
+  - self_attn.o_proj
+lora_modules_to_save:
+  - lm_head
+  - embed_tokens
+
+chat_template: llama4
+datasets:
+  - path: mlabonne/FineTome-100k
+    type: chat_template
+    split: train[:20%]
+    field_messages: conversations
+    message_property_mappings:
+      role: from
+      content: value
+
+dataset_prepared_path: last_run_prepared
+val_set_size: 0.0
+output_dir: ./outputs/out
+
+sequence_len: 4096
+sample_packing: true
+pad_to_sequence_len: true
+
+gradient_accumulation_steps: 1
+micro_batch_size: 1
+num_epochs: 1
+optimizer: adamw_torch_8bit
+lr_scheduler: cosine
+learning_rate: 2e-5
+
+bf16: true
+tf32: true
+
+# gradient_checkpointing: true
+# gradient_checkpointing_kwargs:
+#   use_reentrant: false
+logging_steps: 1
+flash_attention: true
+
+warmup_steps: 100
+evals_per_epoch: 2
+saves_per_epoch: 1
+weight_decay: 0.0
+fsdp:
+  - auto_wrap
+  - full_shard
+fsdp_config:
+  fsdp_version: 2
+  fsdp_offload_params: false
+  fsdp_cpu_ram_efficient_loading: true
+  fsdp_auto_wrap_policy: TRANSFORMER_BASED_WRAP
+  fsdp_transformer_layer_cls_to_wrap: Llama4TextDecoderLayer
+  fsdp_state_dict_type: SHARDED_STATE_DICT
+  fsdp_sharding_strategy: FULL_SHARD
+  fsdp_reshard_after_forward: true
+  fsdp_activation_checkpointing: true
+special_tokens:
+  pad_token: <|finetune_right_pad_id|>
+  eos_token: <|eot|>
@@ -6,18 +6,19 @@ triton>=3.0.0
 mamba-ssm==1.2.0.post1
 xformers>=0.0.23.post1
 autoawq==0.2.7.post3
-liger-kernel==0.5.5
+liger-kernel==0.5.6
 # END section
 
 packaging==23.2
 
-peft==0.15.0
+peft==0.15.1
 transformers==4.51.0
 tokenizers>=0.21.1
 accelerate==1.6.0
 datasets==3.5.0
 deepspeed>=0.15.4
 trl==0.16.1
+hf_xet==1.0.0
 
 optimum==1.16.2
 hf_transfer
 
@@ -562,6 +562,19 @@ def create_accelerator_and_postprocess(self):
 
         return res
 
+    def additional_accelerator_args(
+        self, fp8=None, **kwargs
+    ):  # pylint: disable=unused-argument
+        ret_kwargs = {}
+        if fp8:
+            from accelerate.utils import AORecipeKwargs
+
+            ret_kwargs["mixed_precision"] = "fp8"
+            ret_kwargs["kwargs_handlers"] = [AORecipeKwargs()]
+            os.environ["ACCELERATE_MIXED_PRECISION"] = "fp8"
+
+        return ret_kwargs
+
     def log(self, logs: dict[str, float], start_time: float | None = None) -> None:
         """
         Log `logs` on the various objects watching training, including stored metrics.
 
@@ -173,5 +173,17 @@ def _liger_rms_norm_wrapper(dim, **kwargs):
                 raise NotImplementedError(
                     "Fused linear cross entropy is not yet supported for Gemma3."
                 )
+        elif cfg.model_config_type == "llama4":
+            from axolotl.integrations.liger.models.llama4 import (
+                apply_liger_kernel_to_llama4,
+            )
+
+            apply_liger_kernel_to_llama4(
+                cross_entropy=cfg.liger_cross_entropy,
+                fused_linear_cross_entropy=cfg.liger_fused_linear_cross_entropy,
+                glu_activation=cfg.liger_glu_activation,
+                rms_norm=cfg.liger_rms_norm,
+                layer_norm=cfg.liger_layer_norm,
+            )
         elif cfg.model_config_type in ["deepseek_v3"]:
             raise ValueError(f"Unsupported model config type: {cfg.model_config_type}")
@@ -0,0 +1,171 @@
+"""
+Liger FLCE for llama4
+"""
+
+import sys
+from typing import List, Optional, Tuple, Union
+
+import torch
+from liger_kernel.transformers.model.loss_utils import LigerForCausalLMLoss
+from transformers.modeling_outputs import CausalLMOutputWithPast
+
+
+def lce_forward(
+    self,
+    input_ids: torch.LongTensor = None,
+    attention_mask: Optional[torch.Tensor] = None,
+    position_ids: Optional[torch.LongTensor] = None,
+    past_key_values: Optional[
+        Union["Cache", List[torch.FloatTensor]]  # noqa: F821
+    ] = None,
+    inputs_embeds: Optional[torch.FloatTensor] = None,
+    labels: Optional[torch.LongTensor] = None,
+    use_cache: Optional[bool] = None,
+    output_attentions: Optional[bool] = None,
+    output_hidden_states: Optional[bool] = None,
+    return_dict: Optional[bool] = None,
+    cache_position: Optional[torch.LongTensor] = None,
+    logits_to_keep: Union[int, torch.Tensor] = 0,
+    **loss_kwargs,
+) -> Union[Tuple, CausalLMOutputWithPast]:
+    r"""
+    Args:
+        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
+            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
+            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
+
+        logits_to_keep (`int` or `torch.Tensor`, *optional*):
+            If an `int`, compute logits for the last `logits_to_keep` tokens. If `0`, calculate logits for all
+            `input_ids` (special case). Only last token logits are needed for generation, and calculating them only for that
+            token can save memory, which becomes pretty significant for long sequences or large vocabulary size.
+            If a `torch.Tensor`, must be 1D corresponding to the indices to keep in the sequence length dimension.
+            This is useful when using packed tensor format (single dimension for batch and sequence length).
+
+    Returns:
+    """
+
+    # pylint: disable=duplicate-code
+    output_attentions = (
+        output_attentions
+        if output_attentions is not None
+        else self.config.output_attentions
+    )
+    output_hidden_states = (
+        output_hidden_states
+        if output_hidden_states is not None
+        else self.config.output_hidden_states
+    )
+    return_dict = (
+        return_dict if return_dict is not None else self.config.use_return_dict
+    )
+
+    # decoder outputs consists of (dec_features, layer_state, dec_hidden, dec_attn)
+    outputs = self.model(
+        input_ids=input_ids,
+        attention_mask=attention_mask,
+        position_ids=position_ids,
+        past_key_values=past_key_values,
+        inputs_embeds=inputs_embeds,
+        use_cache=use_cache,
+        output_attentions=output_attentions,
+        output_hidden_states=output_hidden_states,
+        return_dict=return_dict,
+        cache_position=cache_position,
+    )
+
+    hidden_states = outputs[0]
+
+    if hasattr(self.config, "pretraining_tp") and self.config.pretraining_tp > 1:
+        raise Exception(  # pylint: disable=broad-exception-raised
+            "Liger Kernel does not support pretraining_tp!!"
+        )
+
+    logits = None
+    loss = None
+    # if in training mode, don't materialize logits
+    if self.training and (labels is not None):
+        loss = LigerForCausalLMLoss(
+            hidden_states=hidden_states,
+            lm_head_weight=self.lm_head.weight,
+            labels=labels,
+            hidden_size=self.config.hidden_size,
+            **loss_kwargs,
+        )
+
+    else:  # if in inference mode materialize logits
+        slice_indices = (
+            slice(-logits_to_keep, None)
+            if isinstance(logits_to_keep, int)
+            else logits_to_keep
+        )
+        logits = self.lm_head(hidden_states[:, slice_indices, :])
+        if labels is not None:
+            loss = self.loss_function(
+                logits=logits,
+                labels=labels,
+                vocab_size=self.config.vocab_size,
+                **loss_kwargs,
+            )
+
+    if not return_dict:
+        output = (logits,) + outputs[1:]
+        return (loss,) + output if loss is not None else output
+
+    return CausalLMOutputWithPast(
+        loss=loss,
+        logits=logits,
+        past_key_values=outputs.past_key_values,
+        hidden_states=outputs.hidden_states,
+        attentions=outputs.attentions,
+    )
+
+
+def apply_liger_kernel_to_llama4(
+    cross_entropy: bool = False,
+    fused_linear_cross_entropy: bool = False,
+    rms_norm: bool = False,
+    glu_activation: bool = False,
+    layer_norm: bool = False,
+    **kwargs,  # pylint: disable=unused-argument
+) -> None:
+    """
+    Apply Liger kernels to replace original implementation in HuggingFace Llama models (2 and 3)
+
+    Args:
+        cross_entropy (bool): Whether to apply Liger's cross entropy loss. Default is False.
+        fused_linear_cross_entropy (bool):
+            Whether to apply Liger's fused linear cross entropy loss. Default is False.
+            `cross_entropy` and `fused_linear_cross_entropy` cannot both be False.
+            If `fused_linear_cross_entropy` is True, the logits will not be materialized but more memory efficient.
+        rms_norm (bool): Whether to apply Liger's RMSNorm. Default is False.
+        glu_activation (bool): Whether to apply Liger's SwiGLU MLP. Default is False.
+        layer_norm (bool): Whether to apply Liger's LayerNorm. Default is False.
+    """
+
+    import transformers.models.llama4.modeling_llama4  # noqa: F401  # pylint: disable=unused-import
+    from liger_kernel.transformers.functional import liger_cross_entropy
+    from liger_kernel.transformers.layer_norm import LigerLayerNorm
+    from liger_kernel.transformers.rms_norm import LigerRMSNorm
+    from liger_kernel.transformers.swiglu import LigerSwiGLUMLP
+
+    assert not (
+        cross_entropy and fused_linear_cross_entropy
+    ), "cross_entropy and fused_linear_cross_entropy cannot both be True."
+
+    modeling_llama4 = sys.modules["transformers.models.llama4.modeling_llama4"]
+
+    if rms_norm:
+        modeling_llama4.Llama4TextRMSNorm = LigerRMSNorm
+    if glu_activation:
+        modeling_llama4.Llama4TextMLP = LigerSwiGLUMLP
+    if layer_norm:
+        modeling_llama4.nn.LayerNorm = LigerLayerNorm
+
+    if cross_entropy:
+        from transformers.loss.loss_utils import nn
+
+        nn.functional.cross_entropy = liger_cross_entropy
+
+    if fused_linear_cross_entropy:
+        modeling_llama4.Llama4ForCausalLM.forward = lce_forward
@@ -162,7 +162,6 @@ def mask_mod(batch_idx, head_idx, q_idx, kv_idx):
     for n in tuple(sys.modules):
         if ".modeling_" in n and "llama4" not in n:
             if hasattr(sys.modules[n], "make_flex_block_causal_mask"):
-                print(n)
                 sys.modules[n].make_flex_block_causal_mask = (
                     patched_make_flex_block_causal_mask
                 )
 
@@ -13,6 +13,7 @@
 SUPPORTED_MULTIPACK_MODEL_TYPES = [
     "mllama_text_model",
     "llama",
+    "llama4",
     "mistral",
     "mixtral",
     "qwen2",
Original file line number	Diff line number	Diff line change
`@@ -162,7 +162,6 @@ def mask_mod(batch_idx, head_idx, q_idx, kv_idx):`
`162`	`162`	`for n in tuple(sys.modules):`
`163`	`163`	`if ".modeling_" in n and "llama4" not in n:`
`164`	`164`	`if hasattr(sys.modules[n], "make_flex_block_causal_mask"):`
`165`		`- print(n)`
`166`	`165`	`sys.modules[n].make_flex_block_causal_mask = (`
`167`	`166`	`patched_make_flex_block_causal_mask`
`168`	`167`	`)`