huggingface
diff --git a/‎src/diffusers/quantizers/quantization_config.py
Lines changed: 12 additions & 10 deletions b/‎src/diffusers/quantizers/quantization_config.py
Lines changed: 12 additions & 10 deletions
diff --git a/‎src/diffusers/utils/testing_utils.py
Lines changed: 1 addition & 3 deletions b/‎src/diffusers/utils/testing_utils.py
Lines changed: 1 addition & 3 deletions
diff --git a/‎tests/models/autoencoders/test_models_consistency_decoder_vae.py
Lines changed: 3 additions & 2 deletions b/‎tests/models/autoencoders/test_models_consistency_decoder_vae.py
Lines changed: 3 additions & 2 deletions
diff --git a/‎tests/models/unets/test_models_unet_2d.py
Lines changed: 2 additions & 1 deletion b/‎tests/models/unets/test_models_unet_2d.py
Lines changed: 2 additions & 1 deletion
diff --git a/‎tests/models/unets/test_models_unet_2d_condition.py
Lines changed: 2 additions & 3 deletions b/‎tests/models/unets/test_models_unet_2d_condition.py
Lines changed: 2 additions & 3 deletions
diff --git a/‎tests/pipelines/allegro/test_allegro.py
Lines changed: 3 additions & 2 deletions b/‎tests/pipelines/allegro/test_allegro.py
Lines changed: 3 additions & 2 deletions
diff --git a/‎tests/pipelines/audioldm/test_audioldm.py
Lines changed: 5 additions & 5 deletions b/‎tests/pipelines/audioldm/test_audioldm.py
Lines changed: 5 additions & 5 deletions
diff --git a/‎tests/pipelines/audioldm2/test_audioldm2.py
Lines changed: 9 additions & 3 deletions b/‎tests/pipelines/audioldm2/test_audioldm2.py
Lines changed: 9 additions & 3 deletions
diff --git a/‎tests/pipelines/cogvideo/test_cogvideox.py
Lines changed: 3 additions & 2 deletions b/‎tests/pipelines/cogvideo/test_cogvideox.py
Lines changed: 3 additions & 2 deletions
diff --git a/‎tests/pipelines/cogview3/test_cogview3plus.py
Lines changed: 3 additions & 2 deletions b/‎tests/pipelines/cogview3/test_cogview3plus.py
Lines changed: 3 additions & 2 deletions
diff --git a/‎tests/pipelines/controlnet/test_controlnet_img2img.py
Lines changed: 3 additions & 2 deletions b/‎tests/pipelines/controlnet/test_controlnet_img2img.py
Lines changed: 3 additions & 2 deletions
diff --git a/‎tests/pipelines/controlnet/test_controlnet_inpaint.py
Lines changed: 3 additions & 2 deletions b/‎tests/pipelines/controlnet/test_controlnet_inpaint.py
Lines changed: 3 additions & 2 deletions
diff --git a/‎tests/pipelines/controlnet_sd3/test_controlnet_sd3.py
Lines changed: 1 addition & 1 deletion b/‎tests/pipelines/controlnet_sd3/test_controlnet_sd3.py
Lines changed: 1 addition & 1 deletion
diff --git a/‎tests/pipelines/deepfloyd_if/test_if.py
Lines changed: 2 additions & 1 deletion b/‎tests/pipelines/deepfloyd_if/test_if.py
Lines changed: 2 additions & 1 deletion
diff --git a/‎tests/pipelines/deepfloyd_if/test_if_img2img.py
Lines changed: 2 additions & 1 deletion b/‎tests/pipelines/deepfloyd_if/test_if_img2img.py
Lines changed: 2 additions & 1 deletion
diff --git a/‎tests/pipelines/flux/test_pipeline_flux.py
Lines changed: 2 additions & 2 deletions b/‎tests/pipelines/flux/test_pipeline_flux.py
Lines changed: 2 additions & 2 deletions
diff --git a/‎tests/pipelines/flux/test_pipeline_flux_redux.py
Lines changed: 1 addition & 1 deletion b/‎tests/pipelines/flux/test_pipeline_flux_redux.py
Lines changed: 1 addition & 1 deletion
@@ -493,7 +493,7 @@ def __init__(self, quant_type: str, modules_to_not_convert: Optional[List[str]]
         TORCHAO_QUANT_TYPE_METHODS = self._get_torchao_quant_type_to_method()
         if self.quant_type not in TORCHAO_QUANT_TYPE_METHODS.keys():
             is_floating_quant_type = self.quant_type.startswith("float") or self.quant_type.startswith("fp")
-            if is_floating_quant_type and not self._is_cuda_capability_atleast_8_9():
+            if is_floating_quant_type and not self._is_xpu_or_cuda_capability_atleast_8_9():
                 raise ValueError(
                     f"Requested quantization type: {self.quant_type} is not supported on GPUs with CUDA capability <= 8.9. You "
                     f"can check the CUDA capability of your GPU using `torch.cuda.get_device_capability()`."
@@ -645,7 +645,7 @@ def generate_fpx_quantization_types(bits: int):
             QUANTIZATION_TYPES.update(INT8_QUANTIZATION_TYPES)
             QUANTIZATION_TYPES.update(UINTX_QUANTIZATION_DTYPES)
 
-            if cls._is_cuda_capability_atleast_8_9():
+            if cls._is_xpu_or_cuda_capability_atleast_8_9():
                 QUANTIZATION_TYPES.update(FLOATX_QUANTIZATION_TYPES)
 
             return QUANTIZATION_TYPES
@@ -655,14 +655,16 @@ def generate_fpx_quantization_types(bits: int):
             )
 
     @staticmethod
-    def _is_cuda_capability_atleast_8_9() -> bool:
-        if not torch.cuda.is_available():
-            raise RuntimeError("TorchAO requires a CUDA compatible GPU and installation of PyTorch.")
-
-        major, minor = torch.cuda.get_device_capability()
-        if major == 8:
-            return minor >= 9
-        return major >= 9
+    def _is_xpu_or_cuda_capability_atleast_8_9() -> bool:
+        if torch.cuda.is_available():
+            major, minor = torch.cuda.get_device_capability()
+            if major == 8:
+                return minor >= 9
+            return major >= 9
+        elif torch.xpu.is_available():
+            return True
+        else:
+            raise RuntimeError("TorchAO requires a CUDA compatible GPU or Intel XPU and installation of PyTorch.")
 
     def get_apply_tensor_subclass(self):
         TORCHAO_QUANT_TYPE_METHODS = self._get_torchao_quant_type_to_method()
 
@@ -300,9 +300,7 @@ def require_torch_gpu(test_case):
 
 def require_torch_cuda_compatibility(expected_compute_capability):
     def decorator(test_case):
-        if not torch.cuda.is_available():
-            return unittest.skip(test_case)
-        else:
+        if torch.cuda.is_available():
             current_compute_capability = get_torch_cuda_device_capability()
             return unittest.skipUnless(
                 float(current_compute_capability) == float(expected_compute_capability),
 
@@ -21,6 +21,7 @@
 
 from diffusers import ConsistencyDecoderVAE, StableDiffusionPipeline
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     load_image,
     slow,
@@ -162,13 +163,13 @@ def setUp(self):
         # clean up the VRAM before each test
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         # clean up the VRAM after each test
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     @torch.no_grad()
     def test_encode_decode(self):
 
@@ -22,6 +22,7 @@
 from diffusers import UNet2DModel
 from diffusers.utils import logging
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     floats_tensor,
     require_torch_accelerator,
@@ -229,7 +230,7 @@ def test_from_pretrained_accelerate_wont_change_results(self):
 
         # two models don't need to stay in the device at the same time
         del model_accelerate
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
         gc.collect()
 
         model_normal_load, _ = UNet2DModel.from_pretrained(
 
@@ -46,7 +46,6 @@
     require_peft_backend,
     require_torch_accelerator,
     require_torch_accelerator_with_fp16,
-    require_torch_gpu,
     skip_mps,
     slow,
     torch_all_close,
@@ -978,13 +977,13 @@ def test_ip_adapter_plus(self):
         assert sample2.allclose(sample5, atol=1e-4, rtol=1e-4)
         assert sample2.allclose(sample6, atol=1e-4, rtol=1e-4)
 
-    @require_torch_gpu
     @parameterized.expand(
         [
             ("hf-internal-testing/unet2d-sharded-dummy", None),
             ("hf-internal-testing/tiny-sd-unet-sharded-latest-format", "fp16"),
         ]
     )
+    @require_torch_accelerator
     def test_load_sharded_checkpoint_from_hub(self, repo_id, variant):
         _, inputs_dict = self.prepare_init_args_and_inputs_for_common()
         loaded_model = self.model_class.from_pretrained(repo_id, variant=variant)
@@ -994,13 +993,13 @@ def test_load_sharded_checkpoint_from_hub(self, repo_id, variant):
         assert loaded_model
         assert new_output.sample.shape == (4, 4, 16, 16)
 
-    @require_torch_gpu
     @parameterized.expand(
         [
             ("hf-internal-testing/unet2d-sharded-dummy-subfolder", None),
             ("hf-internal-testing/tiny-sd-unet-sharded-latest-format-subfolder", "fp16"),
         ]
     )
+    @require_torch_accelerator
     def test_load_sharded_checkpoint_from_hub_subfolder(self, repo_id, variant):
         _, inputs_dict = self.prepare_init_args_and_inputs_for_common()
         loaded_model = self.model_class.from_pretrained(repo_id, subfolder="unet", variant=variant)
 
@@ -24,6 +24,7 @@
 
 from diffusers import AllegroPipeline, AllegroTransformer3DModel, AutoencoderKLAllegro, DDIMScheduler
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     numpy_cosine_similarity_distance,
     require_hf_hub_version_greater,
@@ -341,12 +342,12 @@ class AllegroPipelineIntegrationTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def test_allegro(self):
         generator = torch.Generator("cpu").manual_seed(0)
 
@@ -37,7 +37,7 @@
     UNet2DConditionModel,
 )
 from diffusers.utils import is_xformers_available
-from diffusers.utils.testing_utils import enable_full_determinism, nightly, torch_device
+from diffusers.utils.testing_utils import backend_empty_cache, enable_full_determinism, nightly, torch_device
 
 from ..pipeline_params import TEXT_TO_AUDIO_BATCH_PARAMS, TEXT_TO_AUDIO_PARAMS
 from ..test_pipelines_common import PipelineTesterMixin
@@ -378,12 +378,12 @@ class AudioLDMPipelineSlowTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def get_inputs(self, device, generator_device="cpu", dtype=torch.float32, seed=0):
         generator = torch.Generator(device=generator_device).manual_seed(seed)
@@ -423,12 +423,12 @@ class AudioLDMPipelineNightlyTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def get_inputs(self, device, generator_device="cpu", dtype=torch.float32, seed=0):
         generator = torch.Generator(device=generator_device).manual_seed(seed)
 
@@ -45,7 +45,13 @@
     LMSDiscreteScheduler,
     PNDMScheduler,
 )
-from diffusers.utils.testing_utils import enable_full_determinism, is_torch_version, nightly, torch_device
+from diffusers.utils.testing_utils import (
+    backend_empty_cache,
+    enable_full_determinism,
+    is_torch_version,
+    nightly,
+    torch_device,
+)
 
 from ..pipeline_params import TEXT_TO_AUDIO_BATCH_PARAMS, TEXT_TO_AUDIO_PARAMS
 from ..test_pipelines_common import PipelineTesterMixin
@@ -540,12 +546,12 @@ class AudioLDM2PipelineSlowTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def get_inputs(self, device, generator_device="cpu", dtype=torch.float32, seed=0):
         generator = torch.Generator(device=generator_device).manual_seed(seed)
 
@@ -22,6 +22,7 @@
 
 from diffusers import AutoencoderKLCogVideoX, CogVideoXPipeline, CogVideoXTransformer3DModel, DDIMScheduler
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     numpy_cosine_similarity_distance,
     require_torch_accelerator,
@@ -334,12 +335,12 @@ class CogVideoXPipelineIntegrationTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def test_cogvideox(self):
         generator = torch.Generator("cpu").manual_seed(0)
 
@@ -22,6 +22,7 @@
 
 from diffusers import AutoencoderKL, CogVideoXDDIMScheduler, CogView3PlusPipeline, CogView3PlusTransformer2DModel
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     numpy_cosine_similarity_distance,
     require_torch_accelerator,
@@ -244,12 +245,12 @@ class CogView3PlusPipelineIntegrationTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def test_cogview3plus(self):
         generator = torch.Generator("cpu").manual_seed(0)
 
@@ -36,6 +36,7 @@
 from diffusers.utils import load_image
 from diffusers.utils.import_utils import is_xformers_available
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     floats_tensor,
     load_numpy,
@@ -412,12 +413,12 @@ class ControlNetImg2ImgPipelineSlowTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def test_canny(self):
         controlnet = ControlNetModel.from_pretrained("lllyasviel/sd-controlnet-canny")
 
@@ -36,6 +36,7 @@
 from diffusers.utils import load_image
 from diffusers.utils.import_utils import is_xformers_available
 from diffusers.utils.testing_utils import (
+    backend_empty_cache,
     enable_full_determinism,
     floats_tensor,
     load_numpy,
@@ -464,12 +465,12 @@ class ControlNetInpaintPipelineSlowTests(unittest.TestCase):
     def setUp(self):
         super().setUp()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def tearDown(self):
         super().tearDown()
         gc.collect()
-        torch.cuda.empty_cache()
+        backend_empty_cache(torch_device)
 
     def test_canny(self):
         controlnet = ControlNetModel.from_pretrained("lllyasviel/sd-controlnet-canny")
 
@@ -221,7 +221,7 @@ def test_xformers_attention_forwardGenerator_pass(self):
 
 @slow
 @require_big_accelerator
-@pytest.mark.big_gpu_with_torch_cuda
+@pytest.mark.big_accelerator
 class StableDiffusion3ControlNetPipelineSlowTests(unittest.TestCase):
     pipeline_class = StableDiffusion3ControlNetPipeline
 
 
@@ -25,6 +25,7 @@
 from diffusers.utils.import_utils import is_xformers_available
 from diffusers.utils.testing_utils import (
     backend_empty_cache,
+    backend_max_memory_allocated,
     backend_reset_max_memory_allocated,
     backend_reset_peak_memory_stats,
     load_numpy,
@@ -135,7 +136,7 @@ def test_if_text_to_image(self):
 
         image = output.images[0]
 
-        mem_bytes = torch.cuda.max_memory_allocated()
+        mem_bytes = backend_max_memory_allocated(torch_device)
         assert mem_bytes < 12 * 10**9
 
         expected_image = load_numpy(
 
@@ -24,6 +24,7 @@
 from diffusers.utils.import_utils import is_xformers_available
 from diffusers.utils.testing_utils import (
     backend_empty_cache,
+    backend_max_memory_allocated,
     backend_reset_max_memory_allocated,
     backend_reset_peak_memory_stats,
     floats_tensor,
@@ -151,7 +152,7 @@ def test_if_img2img(self):
         )
         image = output.images[0]
 
-        mem_bytes = torch.cuda.max_memory_allocated()
+        mem_bytes = backend_max_memory_allocated(torch_device)
         assert mem_bytes < 12 * 10**9
 
         expected_image = load_numpy(
 
@@ -224,7 +224,7 @@ def test_flux_true_cfg(self):
 
 @nightly
 @require_big_accelerator
-@pytest.mark.big_gpu_with_torch_cuda
+@pytest.mark.big_accelerator
 class FluxPipelineSlowTests(unittest.TestCase):
     pipeline_class = FluxPipeline
     repo_id = "black-forest-labs/FLUX.1-schnell"
@@ -312,7 +312,7 @@ def test_flux_inference(self):
 
 @slow
 @require_big_accelerator
-@pytest.mark.big_gpu_with_torch_cuda
+@pytest.mark.big_accelerator
 class FluxIPAdapterPipelineSlowTests(unittest.TestCase):
     pipeline_class = FluxPipeline
     repo_id = "black-forest-labs/FLUX.1-dev"
 
@@ -19,7 +19,7 @@
 
 @slow
 @require_big_accelerator
-@pytest.mark.big_gpu_with_torch_cuda
+@pytest.mark.big_accelerator
 class FluxReduxSlowTests(unittest.TestCase):
     pipeline_class = FluxPriorReduxPipeline
     repo_id = "black-forest-labs/FLUX.1-Redux-dev"
Original file line number	Diff line number	Diff line change
`@@ -46,7 +46,6 @@`
`46`	`46`	`require_peft_backend,`
`47`	`47`	`require_torch_accelerator,`
`48`	`48`	`require_torch_accelerator_with_fp16,`
`49`		`- require_torch_gpu,`
`50`	`49`	`skip_mps,`
`51`	`50`	`slow,`
`52`	`51`	`torch_all_close,`
`@@ -978,13 +977,13 @@ def test_ip_adapter_plus(self):`
`978`	`977`	`assert sample2.allclose(sample5, atol=1e-4, rtol=1e-4)`
`979`	`978`	`assert sample2.allclose(sample6, atol=1e-4, rtol=1e-4)`
`980`	`979`
`981`		`- @require_torch_gpu`
`982`	`980`	`@parameterized.expand(`
`983`	`981`	`[`
`984`	`982`	`("hf-internal-testing/unet2d-sharded-dummy", None),`
`985`	`983`	`("hf-internal-testing/tiny-sd-unet-sharded-latest-format", "fp16"),`
`986`	`984`	`]`
`987`	`985`	`)`
	`986`	`+ @require_torch_accelerator`
`988`	`987`	`def test_load_sharded_checkpoint_from_hub(self, repo_id, variant):`
`989`	`988`	`_, inputs_dict = self.prepare_init_args_and_inputs_for_common()`
`990`	`989`	`loaded_model = self.model_class.from_pretrained(repo_id, variant=variant)`
`@@ -994,13 +993,13 @@ def test_load_sharded_checkpoint_from_hub(self, repo_id, variant):`
`994`	`993`	`assert loaded_model`
`995`	`994`	`assert new_output.sample.shape == (4, 4, 16, 16)`
`996`	`995`
`997`		`- @require_torch_gpu`
`998`	`996`	`@parameterized.expand(`
`999`	`997`	`[`
`1000`	`998`	`("hf-internal-testing/unet2d-sharded-dummy-subfolder", None),`
`1001`	`999`	`("hf-internal-testing/tiny-sd-unet-sharded-latest-format-subfolder", "fp16"),`
`1002`	`1000`	`]`
`1003`	`1001`	`)`
	`1002`	`+ @require_torch_accelerator`
`1004`	`1003`	`def test_load_sharded_checkpoint_from_hub_subfolder(self, repo_id, variant):`
`1005`	`1004`	`_, inputs_dict = self.prepare_init_args_and_inputs_for_common()`
`1006`	`1005`	`loaded_model = self.model_class.from_pretrained(repo_id, subfolder="unet", variant=variant)`