Enable bitsandbytes quantization on warp size 32 AMD GPUs

sstamenk · sstamenk · commit e94363c950ad · 2025-11-16T22:38:51.000+01:00
Signed-off-by: sstamenk &lt;strahinja.stamenkovic@amd.com&gt;
diff --git a/vllm/platforms/rocm.py b/vllm/platforms/rocm.py
@@ -185,6 +185,9 @@ class RocmPlatform(Platform):
         "petit_nvfp4",
         "torchao",
     ]
+    # bitsandbytes is not supported on GPUs with warp size 64 (gfx9)
+    if not on_gfx9():
+        supported_quantization += ["bitsandbytes"]
 
     @classmethod
     def get_vit_attn_backend(

Original file line number	Diff line number	Diff line change
`@@ -185,6 +185,9 @@ class RocmPlatform(Platform):`
`185`	`185`	`"petit_nvfp4",`
`186`	`186`	`"torchao",`
`187`	`187`	`]`
	`188`	`+ # bitsandbytes is not supported on GPUs with warp size 64 (gfx9)`
	`189`	`+ if not on_gfx9():`
	`190`	`+ supported_quantization += ["bitsandbytes"]`
`188`	`191`
`189`	`192`	`@classmethod`
`190`	`193`	`def get_vit_attn_backend(`