SahilCarterr
diff --git a/‎docs/source/en/api/utilities.md‎
Lines changed: 4 additions & 0 deletions b/‎docs/source/en/api/utilities.md‎
Lines changed: 4 additions & 0 deletions
diff --git a/‎docs/source/en/optimization/memory.md‎
Lines changed: 40 additions & 0 deletions b/‎docs/source/en/optimization/memory.md‎
Lines changed: 40 additions & 0 deletions
diff --git a/‎docs/source/en/tutorials/using_peft_for_inference.md‎
Lines changed: 4 additions & 0 deletions b/‎docs/source/en/tutorials/using_peft_for_inference.md‎
Lines changed: 4 additions & 0 deletions
diff --git a/‎src/diffusers/hooks/__init__.py‎
Lines changed: 1 addition & 0 deletions b/‎src/diffusers/hooks/__init__.py‎
Lines changed: 1 addition & 0 deletions
diff --git a/‎src/diffusers/hooks/group_offloading.py‎
Lines changed: 678 additions & 0 deletions b/‎src/diffusers/hooks/group_offloading.py‎
Lines changed: 678 additions & 0 deletions
diff --git a/‎src/diffusers/hooks/layerwise_casting.py‎
Lines changed: 56 additions & 2 deletions b/‎src/diffusers/hooks/layerwise_casting.py‎
Lines changed: 56 additions & 2 deletions
diff --git a/‎src/diffusers/models/autoencoders/autoencoder_oobleck.py‎
Lines changed: 1 addition & 0 deletions b/‎src/diffusers/models/autoencoders/autoencoder_oobleck.py‎
Lines changed: 1 addition & 0 deletions
diff --git a/‎src/diffusers/models/autoencoders/consistency_decoder_vae.py‎
Lines changed: 2 additions & 0 deletions b/‎src/diffusers/models/autoencoders/consistency_decoder_vae.py‎
Lines changed: 2 additions & 0 deletions
diff --git a/‎src/diffusers/models/autoencoders/vq_model.py‎
Lines changed: 1 addition & 0 deletions b/‎src/diffusers/models/autoencoders/vq_model.py‎
Lines changed: 1 addition & 0 deletions
diff --git a/‎src/diffusers/models/modeling_utils.py‎
Lines changed: 91 additions & 1 deletion b/‎src/diffusers/models/modeling_utils.py‎
Lines changed: 91 additions & 1 deletion
@@ -45,3 +45,7 @@ Utility and helper functions for working with 🤗 Diffusers.
 ## apply_layerwise_casting
 
 [[autodoc]] hooks.layerwise_casting.apply_layerwise_casting
+
+## apply_group_offloading
+
+[[autodoc]] hooks.group_offloading.apply_group_offloading
@@ -158,6 +158,46 @@ In order to properly offload models after they're called, it is required to run
 
 </Tip>
 
+## Group offloading
+
+Group offloading is the middle ground between sequential and model offloading. It works by offloading groups of internal layers (either `torch.nn.ModuleList` or `torch.nn.Sequential`), which uses less memory than model-level offloading. It is also faster than sequential-level offloading because the number of device synchronizations is reduced.
+
+To enable group offloading, call the [`~ModelMixin.enable_group_offload`] method on the model if it is a Diffusers model implementation. For any other model implementation, use [`~hooks.group_offloading.apply_group_offloading`]:
+
+```python
+import torch
+from diffusers import CogVideoXPipeline
+from diffusers.hooks import apply_group_offloading
+from diffusers.utils import export_to_video
+
+# Load the pipeline
+onload_device = torch.device("cuda")
+offload_device = torch.device("cpu")
+pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-5b", torch_dtype=torch.bfloat16)
+
+# We can utilize the enable_group_offload method for Diffusers model implementations
+pipe.transformer.enable_group_offload(onload_device=onload_device, offload_device=offload_device, offload_type="leaf_level", use_stream=True)
+
+# For any other model implementations, the apply_group_offloading function can be used
+apply_group_offloading(pipe.text_encoder, onload_device=onload_device, offload_type="block_level", num_blocks_per_group=2)
+apply_group_offloading(pipe.vae, onload_device=onload_device, offload_type="leaf_level")
+
+prompt = (
+    "A panda, dressed in a small, red jacket and a tiny hat, sits on a wooden stool in a serene bamboo forest. "
+    "The panda's fluffy paws strum a miniature acoustic guitar, producing soft, melodic tunes. Nearby, a few other "
+    "pandas gather, watching curiously and some clapping in rhythm. Sunlight filters through the tall bamboo, "
+    "casting a gentle glow on the scene. The panda's face is expressive, showing concentration and joy as it plays. "
+    "The background includes a small, flowing stream and vibrant green foliage, enhancing the peaceful and magical "
+    "atmosphere of this unique musical performance."
+)
+video = pipe(prompt=prompt, guidance_scale=6, num_inference_steps=50).frames[0]
+# This utilized about 14.79 GB. It can be further reduced by using tiling and using leaf_level offloading throughout the pipeline.
+print(f"Max memory reserved: {torch.cuda.max_memory_allocated() / 1024**3:.2f} GB")
+export_to_video(video, "output.mp4", fps=8)
+```
+
+Group offloading (for CUDA devices with support for asynchronous data transfer streams) overlaps data transfer and computation to reduce the overall execution time compared to sequential offloading. This is enabled using layer prefetching with CUDA streams. The next layer to be executed is loaded onto the accelerator device while the current layer is being executed - this increases the memory requirements slightly. Group offloading also supports leaf-level offloading (equivalent to sequential CPU offloading) but can be made much faster when using streams.
+
 ## FP8 layerwise weight-casting
 
 PyTorch supports `torch.float8_e4m3fn` and `torch.float8_e5m2` as weight storage dtypes, but they can't be used for computation in many different tensor operations due to unimplemented kernel support. However, you can use these dtypes to store model weights in fp8 precision and upcast them on-the-fly when the layers are used in the forward pass. This is known as layerwise weight-casting.
 
@@ -221,3 +221,7 @@ pipe.delete_adapters("toy")
 pipe.get_active_adapters()
 ["pixel"]
 ```
+
+## PeftInputAutocastDisableHook
+
+[[autodoc]] hooks.layerwise_casting.PeftInputAutocastDisableHook
@@ -2,6 +2,7 @@
 
 
 if is_torch_available():
+    from .group_offloading import apply_group_offloading
     from .hooks import HookRegistry, ModelHook
     from .layerwise_casting import apply_layerwise_casting, apply_layerwise_casting_hook
     from .pyramid_attention_broadcast import PyramidAttentionBroadcastConfig, apply_pyramid_attention_broadcast
@@ -17,14 +17,16 @@
 
 import torch
 
-from ..utils import get_logger
+from ..utils import get_logger, is_peft_available, is_peft_version
 from .hooks import HookRegistry, ModelHook
 
 
 logger = get_logger(__name__)  # pylint: disable=invalid-name
 
 
 # fmt: off
+_LAYERWISE_CASTING_HOOK = "layerwise_casting"
+_PEFT_AUTOCAST_DISABLE_HOOK = "peft_autocast_disable"
 SUPPORTED_PYTORCH_LAYERS = (
     torch.nn.Conv1d, torch.nn.Conv2d, torch.nn.Conv3d,
     torch.nn.ConvTranspose1d, torch.nn.ConvTranspose2d, torch.nn.ConvTranspose3d,
@@ -34,6 +36,11 @@
 DEFAULT_SKIP_MODULES_PATTERN = ("pos_embed", "patch_embed", "norm", "^proj_in$", "^proj_out$")
 # fmt: on
 
+_SHOULD_DISABLE_PEFT_INPUT_AUTOCAST = is_peft_available() and is_peft_version(">", "0.14.0")
+if _SHOULD_DISABLE_PEFT_INPUT_AUTOCAST:
+    from peft.helpers import disable_input_dtype_casting
+    from peft.tuners.tuners_utils import BaseTunerLayer
+
 
 class LayerwiseCastingHook(ModelHook):
     r"""
@@ -70,6 +77,32 @@ def post_forward(self, module: torch.nn.Module, output):
         return output
 
 
+class PeftInputAutocastDisableHook(ModelHook):
+    r"""
+    A hook that disables the casting of inputs to the module weight dtype during the forward pass. By default, PEFT
+    casts the inputs to the weight dtype of the module, which can lead to precision loss.
+
+    The reasons for needing this are:
+        - If we don't add PEFT layers' weight names to `skip_modules_pattern` when applying layerwise casting, the
+          inputs will be casted to the, possibly lower precision, storage dtype. Reference:
+          https://github.com/huggingface/peft/blob/0facdebf6208139cbd8f3586875acb378813dd97/src/peft/tuners/lora/layer.py#L706
+        - We can, on our end, use something like accelerate's `send_to_device` but for dtypes. This way, we can ensure
+          that the inputs are casted to the computation dtype correctly always. However, there are two goals we are
+          hoping to achieve:
+            1. Making forward implementations independent of device/dtype casting operations as much as possible.
+            2. Peforming inference without losing information from casting to different precisions. With the current
+               PEFT implementation (as linked in the reference above), and assuming running layerwise casting inference
+               with storage_dtype=torch.float8_e4m3fn and compute_dtype=torch.bfloat16, inputs are cast to
+               torch.float8_e4m3fn in the lora layer. We will then upcast back to torch.bfloat16 when we continue the
+               forward pass in PEFT linear forward or Diffusers layer forward, with a `send_to_dtype` operation from
+               LayerwiseCastingHook. This will be a lossy operation and result in poorer generation quality.
+    """
+
+    def new_forward(self, module: torch.nn.Module, *args, **kwargs):
+        with disable_input_dtype_casting(module):
+            return self.fn_ref.original_forward(*args, **kwargs)
+
+
 def apply_layerwise_casting(
     module: torch.nn.Module,
     storage_dtype: torch.dtype,
@@ -134,6 +167,7 @@ def apply_layerwise_casting(
         skip_modules_classes,
         non_blocking,
     )
+    _disable_peft_input_autocast(module)
 
 
 def _apply_layerwise_casting(
@@ -188,4 +222,24 @@ def apply_layerwise_casting_hook(
     """
     registry = HookRegistry.check_if_exists_or_initialize(module)
     hook = LayerwiseCastingHook(storage_dtype, compute_dtype, non_blocking)
-    registry.register_hook(hook, "layerwise_casting")
+    registry.register_hook(hook, _LAYERWISE_CASTING_HOOK)
+
+
+def _is_layerwise_casting_active(module: torch.nn.Module) -> bool:
+    for submodule in module.modules():
+        if (
+            hasattr(submodule, "_diffusers_hook")
+            and submodule._diffusers_hook.get_hook(_LAYERWISE_CASTING_HOOK) is not None
+        ):
+            return True
+    return False
+
+
+def _disable_peft_input_autocast(module: torch.nn.Module) -> None:
+    if not _SHOULD_DISABLE_PEFT_INPUT_AUTOCAST:
+        return
+    for submodule in module.modules():
+        if isinstance(submodule, BaseTunerLayer) and _is_layerwise_casting_active(submodule):
+            registry = HookRegistry.check_if_exists_or_initialize(submodule)
+            hook = PeftInputAutocastDisableHook()
+            registry.register_hook(hook, _PEFT_AUTOCAST_DISABLE_HOOK)
@@ -317,6 +317,7 @@ class AutoencoderOobleck(ModelMixin, ConfigMixin):
     """
 
     _supports_gradient_checkpointing = False
+    _supports_group_offloading = False
 
     @register_to_config
     def __init__(
 
@@ -68,6 +68,8 @@ class ConsistencyDecoderVAE(ModelMixin, ConfigMixin):
         ```
     """
 
+    _supports_group_offloading = False
+
     @register_to_config
     def __init__(
         self,
 
@@ -72,6 +72,7 @@ class VQModel(ModelMixin, ConfigMixin):
     """
 
     _skip_layerwise_casting_patterns = ["quantize"]
+    _supports_group_offloading = False
 
     @register_to_config
     def __init__(
 
@@ -34,7 +34,7 @@
 from typing_extensions import Self
 
 from .. import __version__
-from ..hooks import apply_layerwise_casting
+from ..hooks import apply_group_offloading, apply_layerwise_casting
 from ..quantizers import DiffusersAutoQuantizer, DiffusersQuantizer
 from ..quantizers.quantization_config import QuantizationMethod
 from ..utils import (
@@ -87,7 +87,17 @@
 
 
 def get_parameter_device(parameter: torch.nn.Module) -> torch.device:
+    from ..hooks.group_offloading import _get_group_onload_device
+
+    try:
+        # Try to get the onload device from the group offloading hook
+        return _get_group_onload_device(parameter)
+    except ValueError:
+        pass
+
     try:
+        # If the onload device is not available due to no group offloading hooks, try to get the device
+        # from the first parameter or buffer
         parameters_and_buffers = itertools.chain(parameter.parameters(), parameter.buffers())
         return next(parameters_and_buffers).device
     except StopIteration:
@@ -166,6 +176,7 @@ class ModelMixin(torch.nn.Module, PushToHubMixin):
     _no_split_modules = None
     _keep_in_fp32_modules = None
     _skip_layerwise_casting_patterns = None
+    _supports_group_offloading = True
 
     def __init__(self):
         super().__init__()
@@ -437,6 +448,55 @@ def enable_layerwise_casting(
             self, storage_dtype, compute_dtype, skip_modules_pattern, skip_modules_classes, non_blocking
         )
 
+    def enable_group_offload(
+        self,
+        onload_device: torch.device,
+        offload_device: torch.device = torch.device("cpu"),
+        offload_type: str = "block_level",
+        num_blocks_per_group: Optional[int] = None,
+        non_blocking: bool = False,
+        use_stream: bool = False,
+    ) -> None:
+        r"""
+        Activates group offloading for the current model.
+
+        See [`~hooks.group_offloading.apply_group_offloading`] for more information.
+
+        Example:
+
+            ```python
+            >>> from diffusers import CogVideoXTransformer3DModel
+
+            >>> transformer = CogVideoXTransformer3DModel.from_pretrained(
+            ...     "THUDM/CogVideoX-5b", subfolder="transformer", torch_dtype=torch.bfloat16
+            ... )
+
+            >>> transformer.enable_group_offload(
+            ...     onload_device=torch.device("cuda"),
+            ...     offload_device=torch.device("cpu"),
+            ...     offload_type="leaf_level",
+            ...     use_stream=True,
+            ... )
+            ```
+        """
+        if getattr(self, "enable_tiling", None) is not None and getattr(self, "use_tiling", False) and use_stream:
+            msg = (
+                "Applying group offloading on autoencoders, with CUDA streams, may not work as expected if the first "
+                "forward pass is executed with tiling enabled. Please make sure to either:\n"
+                "1. Run a forward pass with small input shapes.\n"
+                "2. Or, run a forward pass with tiling disabled (can still use small dummy inputs)."
+            )
+            logger.warning(msg)
+        if not self._supports_group_offloading:
+            raise ValueError(
+                f"{self.__class__.__name__} does not support group offloading. Please make sure to set the boolean attribute "
+                f"`_supports_group_offloading` to `True` in the class definition. If you believe this is a mistake, please "
+                f"open an issue at https://github.com/huggingface/diffusers/issues."
+            )
+        apply_group_offloading(
+            self, onload_device, offload_device, offload_type, num_blocks_per_group, non_blocking, use_stream
+        )
+
     def save_pretrained(
         self,
         save_directory: Union[str, os.PathLike],
@@ -1170,6 +1230,8 @@ def from_pretrained(cls, pretrained_model_name_or_path: Optional[Union[str, os.P
     # Adapted from `transformers`.
     @wraps(torch.nn.Module.cuda)
     def cuda(self, *args, **kwargs):
+        from ..hooks.group_offloading import _is_group_offload_enabled
+
         # Checks if the model has been loaded in 4-bit or 8-bit with BNB
         if getattr(self, "quantization_method", None) == QuantizationMethod.BITS_AND_BYTES:
             if getattr(self, "is_loaded_in_8bit", False):
@@ -1182,13 +1244,34 @@ def cuda(self, *args, **kwargs):
                     "Calling `cuda()` is not supported for `4-bit` quantized models with the installed version of bitsandbytes. "
                     f"The current device is `{self.device}`. If you intended to move the model, please install bitsandbytes >= 0.43.2."
                 )
+
+        # Checks if group offloading is enabled
+        if _is_group_offload_enabled(self):
+            logger.warning(
+                f"The module '{self.__class__.__name__}' is group offloaded and moving it using `.cuda()` is not supported."
+            )
+            return self
+
         return super().cuda(*args, **kwargs)
 
     # Adapted from `transformers`.
     @wraps(torch.nn.Module.to)
     def to(self, *args, **kwargs):
+        from ..hooks.group_offloading import _is_group_offload_enabled
+
+        device_arg_or_kwarg_present = any(isinstance(arg, torch.device) for arg in args) or "device" in kwargs
         dtype_present_in_args = "dtype" in kwargs
 
+        # Try converting arguments to torch.device in case they are passed as strings
+        for arg in args:
+            if not isinstance(arg, str):
+                continue
+            try:
+                torch.device(arg)
+                device_arg_or_kwarg_present = True
+            except RuntimeError:
+                pass
+
         if not dtype_present_in_args:
             for arg in args:
                 if isinstance(arg, torch.dtype):
@@ -1213,6 +1296,13 @@ def to(self, *args, **kwargs):
                     "Calling `to()` is not supported for `4-bit` quantized models with the installed version of bitsandbytes. "
                     f"The current device is `{self.device}`. If you intended to move the model, please install bitsandbytes >= 0.43.2."
                 )
+
+        if _is_group_offload_enabled(self) and device_arg_or_kwarg_present:
+            logger.warning(
+                f"The module '{self.__class__.__name__}' is group offloaded and moving it using `.to()` is not supported."
+            )
+            return self
+
         return super().to(*args, **kwargs)
 
     # Taken from `transformers`.