From eaa9f5162fbca2ebcb2682eb861bc7e5510a2b66 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 24 Oct 2023 01:49:05 +0800
Subject: Add CPU fp8 support

Since norm layer need fp32, I only convert the linear operation layer(conv2d/linear)

And TE have some pytorch function not support bf16 amp in CPU. I add a condition to indicate if the autocast is for unet.
---
 modules/devices.py | 6 +++++-
 1 file changed, 5 insertions(+), 1 deletion(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 1d4eb563..0cd2b55d 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -71,6 +71,7 @@ def enable_tf32():
 errors.run(enable_tf32, "Enabling TF32")
 
 cpu: torch.device = torch.device("cpu")
+fp8: bool = False
 device: torch.device = None
 device_interrogate: torch.device = None
 device_gfpgan: torch.device = None
@@ -93,10 +94,13 @@ def cond_cast_float(input):
 nv_rng = None
 
 
-def autocast(disable=False):
+def autocast(disable=False, unet=False):
     if disable:
         return contextlib.nullcontext()
 
+    if unet and fp8 and device==cpu:
+        return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
+
     if dtype == torch.float32 or shared.cmd_opts.precision == "full":
         return contextlib.nullcontext()
 
-- 
cgit v1.2.3


From d4d3134f6d2d232c7bcfa80900a362921e644976 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Sat, 28 Oct 2023 15:24:26 +0800
Subject: ManualCast for 10/16 series gpu

---
 modules/devices.py | 57 ++++++++++++++++++++++++++++++++++++++++++++++++------
 1 file changed, 51 insertions(+), 6 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 0cd2b55d..c05f2b35 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -16,6 +16,23 @@ def has_mps() -> bool:
         return mac_specific.has_mps
 
 
+def cuda_no_autocast(device_id=None) -> bool:
+    if device_id is None:
+        device_id = get_cuda_device_id()
+    return (
+        torch.cuda.get_device_capability(device_id) == (7, 5) 
+        and torch.cuda.get_device_name(device_id).startswith("NVIDIA GeForce GTX 16")
+    )
+
+
+def get_cuda_device_id():
+    return (
+        int(shared.cmd_opts.device_id) 
+        if shared.cmd_opts.device_id is not None and shared.cmd_opts.device_id.isdigit() 
+        else 0
+    ) or torch.cuda.current_device()
+
+
 def get_cuda_device_string():
     if shared.cmd_opts.device_id is not None:
         return f"cuda:{shared.cmd_opts.device_id}"
@@ -60,8 +77,7 @@ def enable_tf32():
 
         # enabling benchmark option seems to enable a range of cards to do fp16 when they otherwise can't
         # see https://github.com/AUTOMATIC1111/stable-diffusion-webui/pull/4407
-        device_id = (int(shared.cmd_opts.device_id) if shared.cmd_opts.device_id is not None and shared.cmd_opts.device_id.isdigit() else 0) or torch.cuda.current_device()
-        if torch.cuda.get_device_capability(device_id) == (7, 5) and torch.cuda.get_device_name(device_id).startswith("NVIDIA GeForce GTX 16"):
+        if cuda_no_autocast():
             torch.backends.cudnn.benchmark = True
 
         torch.backends.cuda.matmul.allow_tf32 = True
@@ -92,15 +108,44 @@ def cond_cast_float(input):
 
 
 nv_rng = None
-
-
-def autocast(disable=False, unet=False):
+patch_module_list = [
+    torch.nn.Linear,
+    torch.nn.Conv2d,
+    torch.nn.MultiheadAttention,
+    torch.nn.GroupNorm,
+    torch.nn.LayerNorm,
+]
+
+@contextlib.contextmanager
+def manual_autocast():
+    def manual_cast_forward(self, *args, **kwargs):
+        org_dtype = next(self.parameters()).dtype
+        self.to(dtype)
+        result = self.org_forward(*args, **kwargs)
+        self.to(org_dtype)
+        return result
+    for module_type in patch_module_list:
+        org_forward = module_type.forward
+        module_type.forward = manual_cast_forward
+        module_type.org_forward = org_forward
+    try:
+        yield None
+    finally:
+        for module_type in patch_module_list:
+            module_type.forward = module_type.org_forward
+
+
+def autocast(disable=False):
+    print(fp8, dtype, shared.cmd_opts.precision, device)
     if disable:
         return contextlib.nullcontext()
 
-    if unet and fp8 and device==cpu:
+    if fp8 and device==cpu:
         return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
 
+    if fp8 and (dtype == torch.float32 or shared.cmd_opts.precision == "full" or cuda_no_autocast()):
+        return manual_autocast()
+
     if dtype == torch.float32 or shared.cmd_opts.precision == "full":
         return contextlib.nullcontext()
 
-- 
cgit v1.2.3


From ddc2a3499b8cd120b4a42358bcd33137ce1d1e75 Mon Sep 17 00:00:00 2001
From: KohakuBlueleaf <apolloyeh0123@gmail.com>
Date: Sat, 28 Oct 2023 16:52:35 +0800
Subject: Add MPS manual cast

---
 modules/devices.py | 6 +++++-
 1 file changed, 5 insertions(+), 1 deletion(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index c05f2b35..d7c905c2 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -121,6 +121,8 @@ def manual_autocast():
     def manual_cast_forward(self, *args, **kwargs):
         org_dtype = next(self.parameters()).dtype
         self.to(dtype)
+        args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
+        kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
         result = self.org_forward(*args, **kwargs)
         self.to(org_dtype)
         return result
@@ -136,7 +138,6 @@ def manual_autocast():
 
 
 def autocast(disable=False):
-    print(fp8, dtype, shared.cmd_opts.precision, device)
     if disable:
         return contextlib.nullcontext()
 
@@ -146,6 +147,9 @@ def autocast(disable=False):
     if fp8 and (dtype == torch.float32 or shared.cmd_opts.precision == "full" or cuda_no_autocast()):
         return manual_autocast()
 
+    if has_mps() and shared.cmd_opts.precision != "full":
+        return manual_autocast()
+
     if dtype == torch.float32 or shared.cmd_opts.precision == "full":
         return contextlib.nullcontext()
 
-- 
cgit v1.2.3


From 598da5cd4928618b166886d3485ce30ce3a43490 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Sun, 19 Nov 2023 15:50:06 +0800
Subject: Use options instead of cmd_args

---
 modules/devices.py | 25 ++++++++++++++-----------
 1 file changed, 14 insertions(+), 11 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index d7c905c2..03e7bdb7 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -20,15 +20,15 @@ def cuda_no_autocast(device_id=None) -> bool:
     if device_id is None:
         device_id = get_cuda_device_id()
     return (
-        torch.cuda.get_device_capability(device_id) == (7, 5) 
+        torch.cuda.get_device_capability(device_id) == (7, 5)
         and torch.cuda.get_device_name(device_id).startswith("NVIDIA GeForce GTX 16")
     )
 
 
 def get_cuda_device_id():
     return (
-        int(shared.cmd_opts.device_id) 
-        if shared.cmd_opts.device_id is not None and shared.cmd_opts.device_id.isdigit() 
+        int(shared.cmd_opts.device_id)
+        if shared.cmd_opts.device_id is not None and shared.cmd_opts.device_id.isdigit()
         else 0
     ) or torch.cuda.current_device()
 
@@ -116,16 +116,19 @@ patch_module_list = [
     torch.nn.LayerNorm,
 ]
 
+
+def manual_cast_forward(self, *args, **kwargs):
+    org_dtype = next(self.parameters()).dtype
+    self.to(dtype)
+    args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
+    kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
+    result = self.org_forward(*args, **kwargs)
+    self.to(org_dtype)
+    return result
+
+
 @contextlib.contextmanager
 def manual_autocast():
-    def manual_cast_forward(self, *args, **kwargs):
-        org_dtype = next(self.parameters()).dtype
-        self.to(dtype)
-        args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
-        kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
-        result = self.org_forward(*args, **kwargs)
-        self.to(org_dtype)
-        return result
     for module_type in patch_module_list:
         org_forward = module_type.forward
         module_type.forward = manual_cast_forward
-- 
cgit v1.2.3


From 043d2edcf6a543f236f1f3cb70ac72e7b3b357b6 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Sun, 19 Nov 2023 15:56:31 +0800
Subject: Better naming

---
 modules/devices.py | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 03e7bdb7..c19a7f40 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -128,7 +128,7 @@ def manual_cast_forward(self, *args, **kwargs):
 
 
 @contextlib.contextmanager
-def manual_autocast():
+def manual_cast():
     for module_type in patch_module_list:
         org_forward = module_type.forward
         module_type.forward = manual_cast_forward
@@ -148,10 +148,10 @@ def autocast(disable=False):
         return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
 
     if fp8 and (dtype == torch.float32 or shared.cmd_opts.precision == "full" or cuda_no_autocast()):
-        return manual_autocast()
+        return manual_cast()
 
     if has_mps() and shared.cmd_opts.precision != "full":
-        return manual_autocast()
+        return manual_cast()
 
     if dtype == torch.float32 or shared.cmd_opts.precision == "full":
         return contextlib.nullcontext()
-- 
cgit v1.2.3


From 5768afc776a66bb94e77a9c1daebeea58fa731d5 Mon Sep 17 00:00:00 2001
From: Aarni Koskela <akx@iki.fi>
Date: Sun, 31 Dec 2023 00:20:30 +0200
Subject: Add utility to inspect a model's parameters (to get dtype/device)

---
 modules/devices.py | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index c956207f..bd6bd579 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -4,6 +4,7 @@ from functools import lru_cache
 
 import torch
 from modules import errors, shared
+from modules.torch_utils import get_param
 
 if sys.platform == "darwin":
     from modules import mac_specific
@@ -131,7 +132,7 @@ patch_module_list = [
 
 
 def manual_cast_forward(self, *args, **kwargs):
-    org_dtype = next(self.parameters()).dtype
+    org_dtype = get_param(self).dtype
     self.to(dtype)
     args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
     kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
-- 
cgit v1.2.3


From a70dfb64a86b9b6d869deffdb0ffebe980365473 Mon Sep 17 00:00:00 2001
From: AUTOMATIC1111 <16777216c@gmail.com>
Date: Sun, 31 Dec 2023 22:38:30 +0300
Subject: change import statements for #14478

---
 modules/devices.py | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index bd6bd579..ff279ac5 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -4,7 +4,7 @@ from functools import lru_cache
 
 import torch
 from modules import errors, shared
-from modules.torch_utils import get_param
+from modules import torch_utils
 
 if sys.platform == "darwin":
     from modules import mac_specific
@@ -132,7 +132,7 @@ patch_module_list = [
 
 
 def manual_cast_forward(self, *args, **kwargs):
-    org_dtype = get_param(self).dtype
+    org_dtype = torch_utils.get_param(self).dtype
     self.to(dtype)
     args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
     kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
-- 
cgit v1.2.3


From 209c26a1cb9e4be357ab3c5e7613caf3cbc26183 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 9 Jan 2024 22:11:44 +0800
Subject: improve efficiency and support more device

---
 modules/devices.py | 60 ++++++++++++++++++++++++++++++++++++++----------------
 1 file changed, 43 insertions(+), 17 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index ff279ac5..6edfb127 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -110,6 +110,7 @@ device_codeformer: torch.device = None
 dtype: torch.dtype = torch.float16
 dtype_vae: torch.dtype = torch.float16
 dtype_unet: torch.dtype = torch.float16
+dtype_inference: torch.dtype = torch.float16
 unet_needs_upcast = False
 
 
@@ -131,21 +132,49 @@ patch_module_list = [
 ]
 
 
-def manual_cast_forward(self, *args, **kwargs):
-    org_dtype = torch_utils.get_param(self).dtype
-    self.to(dtype)
-    args = [arg.to(dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
-    kwargs = {k: v.to(dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
-    result = self.org_forward(*args, **kwargs)
-    self.to(org_dtype)
-    return result
+def manual_cast_forward(target_dtype):
+    def forward_wrapper(self, *args, **kwargs):
+        org_dtype = torch_utils.get_param(self).dtype
+        if not target_dtype == org_dtype == dtype_inference:
+            self.to(target_dtype)
+            args = [
+                arg.to(target_dtype) 
+                if isinstance(arg, torch.Tensor) 
+                else arg 
+                for arg in args
+            ]
+            kwargs = {
+                k: v.to(target_dtype) 
+                if isinstance(v, torch.Tensor) 
+                else v 
+                for k, v in kwargs.items()
+            }
+        
+        result = self.org_forward(*args, **kwargs)
+        self.to(org_dtype)
+        
+        if target_dtype != dtype_inference:
+            if isinstance(result, tuple):
+                result = tuple(
+                    i.to(dtype_inference) 
+                    if isinstance(i, torch.Tensor) 
+                    else i 
+                    for i in result
+                )
+            elif isinstance(result, torch.Tensor):
+                result = result.to(dtype_inference)
+        return result
+    return forward_wrapper
 
 
 @contextlib.contextmanager
-def manual_cast():
+def manual_cast(target_dtype):
     for module_type in patch_module_list:
         org_forward = module_type.forward
-        module_type.forward = manual_cast_forward
+        if module_type == torch.nn.MultiheadAttention and has_xpu():
+            module_type.forward = manual_cast_forward(torch.float32)
+        else:
+            module_type.forward = manual_cast_forward(target_dtype)
         module_type.org_forward = org_forward
     try:
         yield None
@@ -161,15 +190,12 @@ def autocast(disable=False):
     if fp8 and device==cpu:
         return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
 
-    if fp8 and (dtype == torch.float32 or shared.cmd_opts.precision == "full" or cuda_no_autocast()):
-        return manual_cast()
-
-    if has_mps() and shared.cmd_opts.precision != "full":
-        return manual_cast()
-
-    if dtype == torch.float32 or shared.cmd_opts.precision == "full":
+    if dtype == torch.float32 and shared.cmd_opts.precision == "full":
         return contextlib.nullcontext()
 
+    if has_xpu() or has_mps() or cuda_no_autocast():
+        return manual_cast(dtype_inference)
+
     return torch.autocast("cuda")
 
 
-- 
cgit v1.2.3


From 42e6df723c68af775b73c9fa4f43f99345348689 Mon Sep 17 00:00:00 2001
From: KohakuBlueleaf <apolloyeh0123@gmail.com>
Date: Tue, 9 Jan 2024 22:39:39 +0800
Subject: Fix bugs when arg dtype doesn't match

---
 modules/devices.py | 25 ++++++++++---------------
 1 file changed, 10 insertions(+), 15 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 6edfb127..e0574052 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -134,24 +134,19 @@ patch_module_list = [
 
 def manual_cast_forward(target_dtype):
     def forward_wrapper(self, *args, **kwargs):
+        if any(
+            isinstance(arg, torch.Tensor) and arg.dtype != target_dtype
+            for arg in args
+        ):
+            args = [arg.to(target_dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
+            kwargs = {k: v.to(target_dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
+        
         org_dtype = torch_utils.get_param(self).dtype
-        if not target_dtype == org_dtype == dtype_inference:
+        if org_dtype != target_dtype:
             self.to(target_dtype)
-            args = [
-                arg.to(target_dtype) 
-                if isinstance(arg, torch.Tensor) 
-                else arg 
-                for arg in args
-            ]
-            kwargs = {
-                k: v.to(target_dtype) 
-                if isinstance(v, torch.Tensor) 
-                else v 
-                for k, v in kwargs.items()
-            }
-        
         result = self.org_forward(*args, **kwargs)
-        self.to(org_dtype)
+        if org_dtype != target_dtype:
+            self.to(org_dtype)
         
         if target_dtype != dtype_inference:
             if isinstance(result, tuple):
-- 
cgit v1.2.3


From c2c05fcca8f3547783c5440c04ec10cc63c65db5 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 9 Jan 2024 22:53:58 +0800
Subject: linting and debugs

---
 modules/devices.py | 12 ++++++------
 1 file changed, 6 insertions(+), 6 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index e0574052..ad36f656 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -140,20 +140,20 @@ def manual_cast_forward(target_dtype):
         ):
             args = [arg.to(target_dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
             kwargs = {k: v.to(target_dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
-        
+
         org_dtype = torch_utils.get_param(self).dtype
         if org_dtype != target_dtype:
             self.to(target_dtype)
         result = self.org_forward(*args, **kwargs)
         if org_dtype != target_dtype:
             self.to(org_dtype)
-        
+
         if target_dtype != dtype_inference:
             if isinstance(result, tuple):
                 result = tuple(
-                    i.to(dtype_inference) 
-                    if isinstance(i, torch.Tensor) 
-                    else i 
+                    i.to(dtype_inference)
+                    if isinstance(i, torch.Tensor)
+                    else i
                     for i in result
                 )
             elif isinstance(result, torch.Tensor):
@@ -185,7 +185,7 @@ def autocast(disable=False):
     if fp8 and device==cpu:
         return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
 
-    if dtype == torch.float32 and shared.cmd_opts.precision == "full":
+    if dtype == torch.float32:
         return contextlib.nullcontext()
 
     if has_xpu() or has_mps() or cuda_no_autocast():
-- 
cgit v1.2.3


From e00365962b17550a42235d1fbe2ad2c7cc4b8961 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 9 Jan 2024 23:13:34 +0800
Subject: Apply correct inference precision implementation

---
 modules/devices.py | 42 +++++++++++++++++++++++++++++++++---------
 1 file changed, 33 insertions(+), 9 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index ad36f656..9e1f207c 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -132,6 +132,21 @@ patch_module_list = [
 ]
 
 
+def cast_output(result):
+    if isinstance(result, tuple):
+        result = tuple(i.to(dtype_inference) if isinstance(i, torch.Tensor) else i for i in result)
+    elif isinstance(result, torch.Tensor):
+        result = result.to(dtype_inference)
+    return result
+
+
+def autocast_with_cast_output(self, *args, **kwargs):
+    result = self.org_forward(*args, **kwargs)
+    if dtype_inference != dtype:
+        result = cast_output(result)
+    return result
+
+
 def manual_cast_forward(target_dtype):
     def forward_wrapper(self, *args, **kwargs):
         if any(
@@ -149,15 +164,7 @@ def manual_cast_forward(target_dtype):
             self.to(org_dtype)
 
         if target_dtype != dtype_inference:
-            if isinstance(result, tuple):
-                result = tuple(
-                    i.to(dtype_inference)
-                    if isinstance(i, torch.Tensor)
-                    else i
-                    for i in result
-                )
-            elif isinstance(result, torch.Tensor):
-                result = result.to(dtype_inference)
+            result = cast_output(result)
         return result
     return forward_wrapper
 
@@ -178,6 +185,20 @@ def manual_cast(target_dtype):
             module_type.forward = module_type.org_forward
 
 
+@contextlib.contextmanager
+def precision_full_with_autocast(autocast_ctx):
+    for module_type in patch_module_list:
+        org_forward = module_type.forward
+        module_type.forward = autocast_with_cast_output
+        module_type.org_forward = org_forward
+    try:
+        with autocast_ctx:
+            yield None
+    finally:
+        for module_type in patch_module_list:
+            module_type.forward = module_type.org_forward
+
+
 def autocast(disable=False):
     if disable:
         return contextlib.nullcontext()
@@ -191,6 +212,9 @@ def autocast(disable=False):
     if has_xpu() or has_mps() or cuda_no_autocast():
         return manual_cast(dtype_inference)
 
+    if dtype_inference == torch.float32 and dtype != torch.float32:
+        return precision_full_with_autocast(torch.autocast("cuda"))
+
     return torch.autocast("cuda")
 
 
-- 
cgit v1.2.3


From 1fd69655fe340325863cbd7bf5297e034a6a3a0a Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 9 Jan 2024 23:15:05 +0800
Subject: Revert "Apply correct inference precision implementation"

This reverts commit e00365962b17550a42235d1fbe2ad2c7cc4b8961.
---
 modules/devices.py | 42 +++++++++---------------------------------
 1 file changed, 9 insertions(+), 33 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 9e1f207c..ad36f656 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -132,21 +132,6 @@ patch_module_list = [
 ]
 
 
-def cast_output(result):
-    if isinstance(result, tuple):
-        result = tuple(i.to(dtype_inference) if isinstance(i, torch.Tensor) else i for i in result)
-    elif isinstance(result, torch.Tensor):
-        result = result.to(dtype_inference)
-    return result
-
-
-def autocast_with_cast_output(self, *args, **kwargs):
-    result = self.org_forward(*args, **kwargs)
-    if dtype_inference != dtype:
-        result = cast_output(result)
-    return result
-
-
 def manual_cast_forward(target_dtype):
     def forward_wrapper(self, *args, **kwargs):
         if any(
@@ -164,7 +149,15 @@ def manual_cast_forward(target_dtype):
             self.to(org_dtype)
 
         if target_dtype != dtype_inference:
-            result = cast_output(result)
+            if isinstance(result, tuple):
+                result = tuple(
+                    i.to(dtype_inference)
+                    if isinstance(i, torch.Tensor)
+                    else i
+                    for i in result
+                )
+            elif isinstance(result, torch.Tensor):
+                result = result.to(dtype_inference)
         return result
     return forward_wrapper
 
@@ -185,20 +178,6 @@ def manual_cast(target_dtype):
             module_type.forward = module_type.org_forward
 
 
-@contextlib.contextmanager
-def precision_full_with_autocast(autocast_ctx):
-    for module_type in patch_module_list:
-        org_forward = module_type.forward
-        module_type.forward = autocast_with_cast_output
-        module_type.org_forward = org_forward
-    try:
-        with autocast_ctx:
-            yield None
-    finally:
-        for module_type in patch_module_list:
-            module_type.forward = module_type.org_forward
-
-
 def autocast(disable=False):
     if disable:
         return contextlib.nullcontext()
@@ -212,9 +191,6 @@ def autocast(disable=False):
     if has_xpu() or has_mps() or cuda_no_autocast():
         return manual_cast(dtype_inference)
 
-    if dtype_inference == torch.float32 and dtype != torch.float32:
-        return precision_full_with_autocast(torch.autocast("cuda"))
-
     return torch.autocast("cuda")
 
 
-- 
cgit v1.2.3


From 58d5b042cd02f287faabef399134b97d323691f2 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 9 Jan 2024 23:23:40 +0800
Subject: Apply the correct behavior of precision='full'

---
 modules/devices.py | 11 +++++++----
 1 file changed, 7 insertions(+), 4 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index ad36f656..29a270d1 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -185,11 +185,14 @@ def autocast(disable=False):
     if fp8 and device==cpu:
         return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
 
-    if dtype == torch.float32:
-        return contextlib.nullcontext()
-
     if has_xpu() or has_mps() or cuda_no_autocast():
-        return manual_cast(dtype_inference)
+        return manual_cast(dtype)
+
+    if fp8 and dtype_inference == torch.float32:
+        return manual_cast(dtype)
+
+    if dtype == torch.float32 or dtype_inference == torch.float32:
+        return contextlib.nullcontext()
 
     return torch.autocast("cuda")
 
-- 
cgit v1.2.3


From ca671e5d7b9d03227f01e6bcb350032b6d14e722 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Tue, 9 Jan 2024 23:30:55 +0800
Subject: rearrange if-statements for cpu

---
 modules/devices.py | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 29a270d1..0321d12c 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -185,15 +185,15 @@ def autocast(disable=False):
     if fp8 and device==cpu:
         return torch.autocast("cpu", dtype=torch.bfloat16, enabled=True)
 
-    if has_xpu() or has_mps() or cuda_no_autocast():
-        return manual_cast(dtype)
-
     if fp8 and dtype_inference == torch.float32:
         return manual_cast(dtype)
 
     if dtype == torch.float32 or dtype_inference == torch.float32:
         return contextlib.nullcontext()
 
+    if has_xpu() or has_mps() or cuda_no_autocast():
+        return manual_cast(dtype)
+
     return torch.autocast("cuda")
 
 
-- 
cgit v1.2.3


From 0181c1f76b97162c42401f1e6286ae73d8aa6033 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Fri, 19 Jan 2024 00:14:03 +0800
Subject: Fix nested manual cast

---
 modules/devices.py | 6 +++++-
 1 file changed, 5 insertions(+), 1 deletion(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 0321d12c..37028629 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -165,6 +165,8 @@ def manual_cast_forward(target_dtype):
 @contextlib.contextmanager
 def manual_cast(target_dtype):
     for module_type in patch_module_list:
+        if hasattr(module_type, "org_forward"):
+            continue
         org_forward = module_type.forward
         if module_type == torch.nn.MultiheadAttention and has_xpu():
             module_type.forward = manual_cast_forward(torch.float32)
@@ -175,7 +177,9 @@ def manual_cast(target_dtype):
         yield None
     finally:
         for module_type in patch_module_list:
-            module_type.forward = module_type.org_forward
+            if hasattr(module_type, "org_forward"):
+                module_type.forward = module_type.org_forward
+                delattr(module_type, "org_forward")
 
 
 def autocast(disable=False):
-- 
cgit v1.2.3


From 81126027f5226e7ee58e1a99194eb9ec7b8ec6e7 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Sat, 20 Jan 2024 16:31:12 +0800
Subject: Avoid early disable

---
 modules/devices.py | 4 ++++
 1 file changed, 4 insertions(+)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 37028629..3bde1699 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -164,9 +164,11 @@ def manual_cast_forward(target_dtype):
 
 @contextlib.contextmanager
 def manual_cast(target_dtype):
+    applied = False
     for module_type in patch_module_list:
         if hasattr(module_type, "org_forward"):
             continue
+        applied = True
         org_forward = module_type.forward
         if module_type == torch.nn.MultiheadAttention and has_xpu():
             module_type.forward = manual_cast_forward(torch.float32)
@@ -176,6 +178,8 @@ def manual_cast(target_dtype):
     try:
         yield None
     finally:
+        if not applied:
+            return
         for module_type in patch_module_list:
             if hasattr(module_type, "org_forward"):
                 module_type.forward = module_type.org_forward
-- 
cgit v1.2.3


From 4a66d2fb228584bb38dc22db6a3e657561834c7a Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Sat, 20 Jan 2024 16:33:59 +0800
Subject: Avoid exceptions to be silenced

---
 modules/devices.py | 11 +++++------
 1 file changed, 5 insertions(+), 6 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 3bde1699..dfffaf24 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -178,12 +178,11 @@ def manual_cast(target_dtype):
     try:
         yield None
     finally:
-        if not applied:
-            return
-        for module_type in patch_module_list:
-            if hasattr(module_type, "org_forward"):
-                module_type.forward = module_type.org_forward
-                delattr(module_type, "org_forward")
+        if applied:
+            for module_type in patch_module_list:
+                if hasattr(module_type, "org_forward"):
+                    module_type.forward = module_type.org_forward
+                    delattr(module_type, "org_forward")
 
 
 def autocast(disable=False):
-- 
cgit v1.2.3


From 750dd6014a45397979cad42a74634451d0861581 Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Mon, 29 Jan 2024 22:27:53 +0800
Subject: Fix potential bugs

---
 modules/devices.py | 9 +++++++--
 1 file changed, 7 insertions(+), 2 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index dfffaf24..60f7d6d7 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -141,7 +141,12 @@ def manual_cast_forward(target_dtype):
             args = [arg.to(target_dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
             kwargs = {k: v.to(target_dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
 
-        org_dtype = torch_utils.get_param(self).dtype
+        org_dtype = target_dtype
+        for param in self.parameters():
+            if param.dtype != target_dtype:
+                org_dtype = param.dtype
+                break
+
         if org_dtype != target_dtype:
             self.to(target_dtype)
         result = self.org_forward(*args, **kwargs)
@@ -170,7 +175,7 @@ def manual_cast(target_dtype):
             continue
         applied = True
         org_forward = module_type.forward
-        if module_type == torch.nn.MultiheadAttention and has_xpu():
+        if module_type == torch.nn.MultiheadAttention:
             module_type.forward = manual_cast_forward(torch.float32)
         else:
             module_type.forward = manual_cast_forward(target_dtype)
-- 
cgit v1.2.3


From 6e7f0860f7ae4a0ce59f9416fb9b2f3bcab44f1d Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Mon, 29 Jan 2024 22:46:43 +0800
Subject: linting

---
 modules/devices.py | 1 -
 1 file changed, 1 deletion(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 60f7d6d7..8f49f7a4 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -4,7 +4,6 @@ from functools import lru_cache
 
 import torch
 from modules import errors, shared
-from modules import torch_utils
 
 if sys.platform == "darwin":
     from modules import mac_specific
-- 
cgit v1.2.3


From d243e24f539d717b221992e894a5db5a321bf3cd Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Mon, 29 Jan 2024 22:49:45 +0800
Subject: Try to reverse the dtype checking mechanism

---
 modules/devices.py | 7 ++-----
 1 file changed, 2 insertions(+), 5 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index 8f49f7a4..f9648e9a 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -4,6 +4,7 @@ from functools import lru_cache
 
 import torch
 from modules import errors, shared
+from modules import torch_utils
 
 if sys.platform == "darwin":
     from modules import mac_specific
@@ -140,11 +141,7 @@ def manual_cast_forward(target_dtype):
             args = [arg.to(target_dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
             kwargs = {k: v.to(target_dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
 
-        org_dtype = target_dtype
-        for param in self.parameters():
-            if param.dtype != target_dtype:
-                org_dtype = param.dtype
-                break
+        org_dtype = torch_utils.get_param(self).dtype
 
         if org_dtype != target_dtype:
             self.to(target_dtype)
-- 
cgit v1.2.3


From f9ba7e648ad5bf7dbdf2b95fa207936179bf784e Mon Sep 17 00:00:00 2001
From: Kohaku-Blueleaf <59680068+KohakuBlueleaf@users.noreply.github.com>
Date: Mon, 29 Jan 2024 22:54:12 +0800
Subject: Revert "Try to reverse the dtype checking mechanism"

This reverts commit d243e24f539d717b221992e894a5db5a321bf3cd.
---
 modules/devices.py | 7 +++++--
 1 file changed, 5 insertions(+), 2 deletions(-)

(limited to 'modules/devices.py')

diff --git a/modules/devices.py b/modules/devices.py
index f9648e9a..8f49f7a4 100644
--- a/modules/devices.py
+++ b/modules/devices.py
@@ -4,7 +4,6 @@ from functools import lru_cache
 
 import torch
 from modules import errors, shared
-from modules import torch_utils
 
 if sys.platform == "darwin":
     from modules import mac_specific
@@ -141,7 +140,11 @@ def manual_cast_forward(target_dtype):
             args = [arg.to(target_dtype) if isinstance(arg, torch.Tensor) else arg for arg in args]
             kwargs = {k: v.to(target_dtype) if isinstance(v, torch.Tensor) else v for k, v in kwargs.items()}
 
-        org_dtype = torch_utils.get_param(self).dtype
+        org_dtype = target_dtype
+        for param in self.parameters():
+            if param.dtype != target_dtype:
+                org_dtype = param.dtype
+                break
 
         if org_dtype != target_dtype:
             self.to(target_dtype)
-- 
cgit v1.2.3