Merge pull request #14597 from AUTOMATIC1111/improved-manual-cast

Improve the implementation of Manual Cast and IPEX support

Merge pull request #14597 from AUTOMATIC1111/improved-manual-cast
Improve the implementation of Manual Cast and IPEX support
905b1423 · AUTOMATIC1111 · GitHub · 6869d958 · ca671e5d · 905b1423
Commit 905b1423 authored Jan 09, 2024 by AUTOMATIC1111 Committed by GitHub Jan 09, 2024
Hide whitespace changes
Inline Side-by-side

Showing with 41 additions and 16 deletions

modules/devices.py modules/devices.py +40 -16

modules/shared_init.py modules/shared_init.py +1 -0

No files found.
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -110,6 +110,7 @@ device_codeformer: torch.device = None
 dtype: torch.dtype = torch.float16
 dtype_vae: torch.dtype = torch.float16
 dtype_unet: torch.dtype = torch.float16
+dtype_inference: torch.dtype = torch.float16
 unet_needs_upcast = False
@@ -131,21 +132,44 @@ patch_module_list = [
 ]
-def manual_cast_forward(self, *args, **kwargs):
+def manual_cast_forward(target_dtype):
-    org_dtype = torch_utils.get_param(self).dtype
+    def forward_wrapper(self, *args, **kwargs):
-    self.to(dtype)
+        if any(
-    args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
+            isinstance(arg, torch.Tensor) and arg.dtype != target_dtype
-    kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
+            for arg in args
-    result = self.org_forward(*args, **kwargs)
+        ):
-    self.to(org_dtype)
+            args = [arg.to(target_dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
-    return result
+            kwargs = {k: v.to(target_dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
+        org_dtype = torch_utils.get_param(self).dtype
+        if org_dtype != target_dtype:
+            self.to(target_dtype)
+        result = self.org_forward(*args, **kwargs)
+        if org_dtype != target_dtype:
+            self.to(org_dtype)
+        if target_dtype != dtype_inference:
+            if isinstance(result, tuple):
+                result = tuple(
+                    i.to(dtype_inference)
+                    if isinstance(i, torch.Tensor)
+                    else i
+                    for i in result
+                )
+            elif isinstance(result, torch.Tensor):
+                result = result.to(dtype_inference)
+        return result
+    return forward_wrapper
 @contextlib.contextmanager
-def manual_cast():
+def manual_cast(target_dtype):
    for module_type in patch_module_list:
        org_forward = module_type.forward
-        module_type.forward = manual_cast_forward
+        if module_type == torch.nn.MultiheadAttention and has_xpu():
+            module_type.forward = manual_cast_forward(torch.float32)
+        else:
+            module_type.forward = manual_cast_forward(target_dtype)
        module_type.org_forward = org_forward
    try:
        yield None
@@ -161,15 +185,15 @@ def autocast(disable=False):
    if fp8 and device==cpu:
        return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
-    if fp8 and (dtype == torch.float32 or shared.cmd_opts.precision == "full" or cuda_no_autocast()):
+    if fp8 and dtype_inference == torch.float32:
-        return manual_cast()
+        return manual_cast(dtype)
-    if has_mps() and shared.cmd_opts.precision != "full":
+    if dtype == torch.float32 or dtype_inference == torch.float32:
-        return manual_cast()
-    if dtype == torch.float32 or shared.cmd_opts.precision == "full":
        return contextlib.nullcontext()
+    if has_xpu() or has_mps() or cuda_no_autocast():
+        return manual_cast(dtype)
    return torch.autocast("cuda")

--- a/modules/shared_init.py
+++ b/modules/shared_init.py
@@ -29,6 +29,7 @@ def initialize():
    devices.dtype = torch.float32 if cmd_opts.no_half else torch.float16
    devices.dtype_vae = torch.float32 if cmd_opts.no_half or cmd_opts.no_half_vae else torch.float16
+    devices.dtype_inference = torch.float32 if cmd_opts.precision == 'full' else devices.dtype
    shared.device = devices.device
    shared.weight_load_location = None if cmd_opts.lowram else "cpu"