Merge 90d14b9eb0 into fa53f71ec0

Remove priority
Add Flash, cuDNN, Efficient attention for Flux
2026-04-15 00:32:25 +00:00 · 2026-04-05 00:39:06 +00:00 · 2025-04-12 04:09:39 -04:00 · 2025-04-11 23:14:41 -04:00
4 changed files with 12 additions and 18 deletions
--- a/README-ja.md
+++ b/README-ja.md
@@ -50,9 +50,6 @@ Stable Diffusion等の画像生成モデルの学習、モデルによる画像

 ### 更新履歴

- 次のリリースに含まれる予定の主な変更点は以下の通りです。リリース前の変更点は予告なく変更される可能性があります。
-    - Intel GPUの互換性を向上しました。[PR #2307](https://github.com/kohya-ss/sd-scripts/pull/2307) WhitePr氏に感謝します。
-
 - **Version 0.10.3 (2026-04-02):**
    - Animaでfp16で学習する際の安定性をさらに改善しました。[PR #2302](https://github.com/kohya-ss/sd-scripts/pull/2302) 問題をご報告いただいた方々に深く感謝します。

--- a/README.md
+++ b/README.md
@@ -47,9 +47,6 @@ If you find this project helpful, please consider supporting its development via

 ### Change History

- The following are the main changes planned for the next release. Please note that these changes may be subject to change without notice before the release.
-    - Improved compatibility with Intel GPUs. Thanks to WhitePr for [PR #2307](https://github.com/kohya-ss/sd-scripts/pull/2307).
-
 - **Version 0.10.3 (2026-04-02):**
    - Stability when training with fp16 on Anima has been further improved. See [PR #2302](https://github.com/kohya-ss/sd-scripts/pull/2302) for details. We deeply appreciate those who reported the issue.

--- a/library/flux_models.py
+++ b/library/flux_models.py
@@ -18,6 +18,7 @@ import torch
 from einops import rearrange
 from torch import Tensor, nn
 from torch.utils.checkpoint import checkpoint
+from torch.nn.attention import SDPBackend, sdpa_kernel

 from library import custom_offloading_utils

@@ -445,11 +446,13 @@ configs = {

 # region math

+kernels = [SDPBackend.FLASH_ATTENTION, SDPBackend.CUDNN_ATTENTION, SDPBackend.EFFICIENT_ATTENTION, SDPBackend.MATH]

 def attention(q: Tensor, k: Tensor, v: Tensor, pe: Tensor, attn_mask: Optional[Tensor] = None) -> Tensor:
    q, k = apply_rope(q, k, pe)

-    x = torch.nn.functional.scaled_dot_product_attention(q, k, v, attn_mask=attn_mask)
+    with sdpa_kernel(kernels):
+        x = torch.nn.functional.scaled_dot_product_attention(q, k, v, attn_mask=attn_mask)
    x = rearrange(x, "B H L D -> B L (H D)")

    return x
--- a/library/ipex/init.py
+++ b/library/ipex/init.py
@@ -1,7 +1,6 @@
 import os
 import sys
 import torch
-from packaging import version
 try:
    import intel_extension_for_pytorch as ipex # pylint: disable=import-error, unused-import
    has_ipex = True
@@ -9,7 +8,7 @@ except Exception:
    has_ipex = False
 from .hijacks import ipex_hijacks

-torch_version = version.parse(torch.__version__)
+torch_version = float(torch.__version__[:3])

 # pylint: disable=protected-access, missing-function-docstring, line-too-long

@@ -57,6 +56,7 @@ def ipex_init(): # pylint: disable=too-many-statements
            torch.cuda.__path__ = torch.xpu.__path__
            torch.cuda.set_stream = torch.xpu.set_stream
            torch.cuda.torch = torch.xpu.torch
+            torch.cuda.Union = torch.xpu.Union
            torch.cuda.__annotations__ = torch.xpu.__annotations__
            torch.cuda.__package__ = torch.xpu.__package__
            torch.cuda.__builtins__ = torch.xpu.__builtins__
@@ -64,12 +64,14 @@ def ipex_init(): # pylint: disable=too-many-statements
            torch.cuda.StreamContext = torch.xpu.StreamContext
            torch.cuda._lazy_call = torch.xpu._lazy_call
            torch.cuda.random = torch.xpu.random
+            torch.cuda._device = torch.xpu._device
            torch.cuda.__name__ = torch.xpu.__name__
+            torch.cuda._device_t = torch.xpu._device_t
            torch.cuda.__spec__ = torch.xpu.__spec__
            torch.cuda.__file__ = torch.xpu.__file__
            # torch.cuda.is_current_stream_capturing = torch.xpu.is_current_stream_capturing

-            if torch_version < version.parse("2.3"):
+            if torch_version < 2.3:
                torch.cuda._initialization_lock = torch.xpu.lazy_init._initialization_lock
                torch.cuda._initialized = torch.xpu.lazy_init._initialized
                torch.cuda._is_in_bad_fork = torch.xpu.lazy_init._is_in_bad_fork
@@ -112,22 +114,17 @@ def ipex_init(): # pylint: disable=too-many-statements
                torch.cuda.threading = torch.xpu.threading
                torch.cuda.traceback = torch.xpu.traceback

-            if torch_version < version.parse("2.5"):
+            if torch_version < 2.5:
                torch.cuda.os = torch.xpu.os
                torch.cuda.Device = torch.xpu.Device
                torch.cuda.warnings = torch.xpu.warnings
                torch.cuda.classproperty = torch.xpu.classproperty
                torch.UntypedStorage.cuda = torch.UntypedStorage.xpu

-            if torch_version < version.parse("2.7"):
+            if torch_version < 2.7:
                torch.cuda.Tuple = torch.xpu.Tuple
                torch.cuda.List = torch.xpu.List

-            if torch_version < version.parse("2.11"):
-                torch.cuda._device_t = torch.xpu._device_t
-                torch.cuda._device = torch.xpu._device
-                torch.cuda.Union = torch.xpu.Union
-

            # Memory:
            if 'linux' in sys.platform and "WSL2" in os.popen("uname -a").read():
@@ -163,7 +160,7 @@ def ipex_init(): # pylint: disable=too-many-statements
            torch.cuda.initial_seed = torch.xpu.initial_seed

            # C
-            if torch_version < version.parse("2.3"):
+            if torch_version < 2.3:
                torch._C._cuda_getCurrentRawStream = ipex._C._getCurrentRawStream
                ipex._C._DeviceProperties.multi_processor_count = ipex._C._DeviceProperties.gpu_subslice_count
                ipex._C._DeviceProperties.major = 12
Author	SHA1	Message	Date
Dave Lage	12ea9b2ec5	Merge `90d14b9eb0` into `fa53f71ec0`	2026-04-05 00:39:06 +00:00
rockerBOO	90d14b9eb0	Remove priority	2025-04-12 04:09:39 -04:00
rockerBOO	2ab5bc69e6	Add Flash, cuDNN, Efficient attention for Flux	2025-04-11 23:14:41 -04:00