Enable bitsandbytes quantization on warp size 32 AMD GPUs

sstamenk · sstamenk · commit 6a06234a4574 · 2025-10-23T13:35:00.000+02:00
Signed-off-by: sstamenk &lt;strahinja.stamenkovic@amd.com&gt;
diff --git a/vllm/platforms/rocm.py b/vllm/platforms/rocm.py
@@ -202,6 +202,9 @@ class RocmPlatform(Platform):
         "petit_nvfp4",
         "torchao",
     ]
+    # bitsandbytes is not supported on GPUs with warp size 64 (gfx9)
+    if not on_gfx9():
+        supported_quantization += ["bitsandbytes"]
 
     @classmethod
     def get_vit_attn_backend(cls, head_size: int, dtype: torch.dtype) -> "_Backend":

Original file line number	Diff line number	Diff line change
`@@ -202,6 +202,9 @@ class RocmPlatform(Platform):`
`202`	`202`	`"petit_nvfp4",`
`203`	`203`	`"torchao",`
`204`	`204`	`]`
	`205`	`+ # bitsandbytes is not supported on GPUs with warp size 64 (gfx9)`
	`206`	`+ if not on_gfx9():`
	`207`	`+ supported_quantization += ["bitsandbytes"]`
`205`	`208`
`206`	`209`	`@classmethod`
`207`	`210`	`def get_vit_attn_backend(cls, head_size: int, dtype: torch.dtype) -> "_Backend":`