Disable ipex autocast due to its bad perf

2026-03-13 17:10:23 +00:00 · 2023-12-02 14:00:46 +08:00
parent 8b40f475a3
commit 7499148ad4
4 changed files with 51 additions and 17 deletions
--- a/modules/cmd_args.py
+++ b/modules/cmd_args.py
@@ -70,6 +70,7 @@ parser.add_argument("--opt-sdp-no-mem-attention", action='store_true', help="pre
 parser.add_argument("--disable-opt-split-attention", action='store_true', help="prefer no cross-attention layer optimization for automatic choice of optimization")
 parser.add_argument("--disable-nan-check", action='store_true', help="do not check if produced images/latent spaces have nans; useful for running without a checkpoint in CI")
 parser.add_argument("--use-cpu", nargs='+', help="use CPU as torch device for specified modules", default=[], type=str.lower)
+parser.add_argument("--use-ipex", action="store_true", help="use Intel XPU as torch device")
 parser.add_argument("--disable-model-loading-ram-optimization", action='store_true', help="disable an optimization that reduces RAM use when loading a model")
 parser.add_argument("--listen", action='store_true', help="launch gradio with 0.0.0.0 as server name, allowing to respond to network requests")
 parser.add_argument("--port", type=int, help="launch gradio with given server port, you need root/admin rights for ports < 1024, defaults to 7860 if available", default=None)
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -3,11 +3,18 @@ import contextlib
 from functools import lru_cache

 import torch
-from modules import errors, shared, xpu_specific
+from modules import errors, shared

 if sys.platform == "darwin":
    from modules import mac_specific

+if shared.cmd_opts.use_ipex:
+    from modules import xpu_specific
+
+
+def has_xpu() -> bool:
+    return shared.cmd_opts.use_ipex and xpu_specific.has_xpu
+

 def has_mps() -> bool:
    if sys.platform != "darwin":
@@ -30,7 +37,7 @@ def get_optimal_device_name():
    if has_mps():
        return "mps"

-    if xpu_specific.has_ipex:
+    if has_xpu():
        return xpu_specific.get_xpu_device_string()

    return "cpu"
@@ -57,6 +64,9 @@ def torch_gc():
    if has_mps():
        mac_specific.torch_mps_gc()

+    if has_xpu():
+        xpu_specific.torch_xpu_gc()
+

 def enable_tf32():
    if torch.cuda.is_available():
@@ -103,15 +113,11 @@ def autocast(disable=False):
    if dtype == torch.float32 or shared.cmd_opts.precision == "full":
        return contextlib.nullcontext()

-    if xpu_specific.has_xpu:
-        return torch.autocast("xpu")
-
    return torch.autocast("cuda")


 def without_autocast(disable=False):
-    device_type = "xpu" if xpu_specific.has_xpu else "cuda"
-    return torch.autocast(device_type, enabled=False) if torch.is_autocast_enabled() and not disable else contextlib.nullcontext()
+    return torch.autocast("cuda", enabled=False) if torch.is_autocast_enabled() and not disable else contextlib.nullcontext()


 class NansException(Exception):
--- a/modules/xpu_specific.py
+++ b/modules/xpu_specific.py
@@ -1,4 +1,3 @@
-import contextlib
 from modules import shared
 from modules.sd_hijack_utils import CondFunc

@@ -10,33 +9,42 @@ try:
 except Exception:
    pass

+
 def check_for_xpu():
-    if not has_ipex:
-        return False
+    return has_ipex and hasattr(torch, 'xpu') and torch.xpu.is_available()

-    return hasattr(torch, 'xpu') and torch.xpu.is_available()
-
-has_xpu = check_for_xpu()

 def get_xpu_device_string():
    if shared.cmd_opts.device_id is not None:
        return f"xpu:{shared.cmd_opts.device_id}"
    return "xpu"

-def return_null_context(*args, **kwargs): # pylint: disable=unused-argument
-    return contextlib.nullcontext()
+
+def torch_xpu_gc():
+    with torch.xpu.device(get_xpu_device_string()):
+        torch.xpu.empty_cache()
+
+
+has_xpu = check_for_xpu()

 if has_xpu:
+    # W/A for https://github.com/intel/intel-extension-for-pytorch/issues/452: torch.Generator API doesn't support XPU device
    CondFunc('torch.Generator',
        lambda orig_func, device=None: torch.xpu.Generator(device),
-        lambda orig_func, device=None: device is not None and device != torch.device("cpu") and device != "cpu")
+        lambda orig_func, device=None: device is not None and device.type == "xpu")

+    # W/A for some OPs that could not handle different input dtypes
    CondFunc('torch.nn.functional.layer_norm',
        lambda orig_func, input, normalized_shape=None, weight=None, *args, **kwargs:
        orig_func(input.to(weight.data.dtype), normalized_shape, weight, *args, **kwargs),
        lambda orig_func, input, normalized_shape=None, weight=None, *args, **kwargs:
        weight is not None and input.dtype != weight.data.dtype)
-
    CondFunc('torch.nn.modules.GroupNorm.forward',
        lambda orig_func, self, input: orig_func(self, input.to(self.weight.data.dtype)),
        lambda orig_func, self, input: input.dtype != self.weight.data.dtype)
+    CondFunc('torch.nn.modules.linear.Linear.forward',
+        lambda orig_func, self, input: orig_func(self, input.to(self.weight.data.dtype)),
+        lambda orig_func, self, input: input.dtype != self.weight.data.dtype)
+    CondFunc('torch.nn.modules.conv.Conv2d.forward',
+        lambda orig_func, self, input: orig_func(self, input.to(self.weight.data.dtype)),
+        lambda orig_func, self, input: input.dtype != self.weight.data.dtype)