Merge c19c90d66e into 51435f1718

Merge pull request #2303 from kohya-ss/sd3
fix: improve numerical stability by conditionally using float32 in Anima with fp16 training
2026-04-18 01:30:02 +00:00 · 2026-04-05 02:56:20 +00:00 · 2026-04-02 12:40:48 +09:00 · 2026-04-02 12:36:29 +09:00 · 2025-10-05 15:33:41 +03:00 · 2025-09-28 15:10:09 +03:00
4 changed files with 49 additions and 4 deletions
--- a/README-ja.md
+++ b/README-ja.md
@@ -50,6 +50,9 @@ Stable Diffusion等の画像生成モデルの学習、モデルによる画像
 ### 更新履歴
 - **Version 0.10.3 (2026-04-02):**
    - Animaでfp16で学習する際の安定性をさらに改善しました。[PR #2302](https://github.com/kohya-ss/sd-scripts/pull/2302) 問題をご報告いただいた方々に深く感謝します。
 - **Version 0.10.2 (2026-03-30):**
    - SD/SDXLのLECO学習に対応しました。[PR #2285](https://github.com/kohya-ss/sd-scripts/pull/2285) および [PR #2294](https://github.com/kohya-ss/sd-scripts/pull/2294) umisetokikaze氏に深く感謝します。
        - 詳細は[ドキュメント](./docs/train_leco.md)をご覧ください。
--- a/README.md
+++ b/README.md
@@ -47,6 +47,9 @@ If you find this project helpful, please consider supporting its development via
 ### Change History
 - **Version 0.10.3 (2026-04-02):**
    - Stability when training with fp16 on Anima has been further improved. See [PR #2302](https://github.com/kohya-ss/sd-scripts/pull/2302) for details. We deeply appreciate those who reported the issue.
 - **Version 0.10.2 (2026-03-30):**
    - LECO training for SD/SDXL is now supported. Many thanks to umisetokikaze for [PR #2285](https://github.com/kohya-ss/sd-scripts/pull/2285) and [PR #2294](https://github.com/kohya-ss/sd-scripts/pull/2294).
        - Please refer to the [documentation](./docs/train_leco.md) for details.
--- a/library/anima_models.py
+++ b/library/anima_models.py
@@ -738,9 +738,9 @@ class FinalLayer(nn.Module):
        x_B_T_H_W_D: torch.Tensor,
        emb_B_T_D: torch.Tensor,
        adaln_lora_B_T_3D: Optional[torch.Tensor] = None,
        use_fp32: bool = False,
    ):
        # Compute AdaLN modulation parameters (in float32 when fp16 to avoid overflow in Linear layers)
        use_fp32 = x_B_T_H_W_D.dtype == torch.float16
        with torch.autocast(device_type=x_B_T_H_W_D.device.type, dtype=torch.float32, enabled=use_fp32):
            if self.use_adaln_lora:
                assert adaln_lora_B_T_3D is not None
@@ -863,11 +863,11 @@ class Block(nn.Module):
        emb_B_T_D: torch.Tensor,
        crossattn_emb: torch.Tensor,
        attn_params: attention.AttentionParams,
        use_fp32: bool = False,
        rope_emb_L_1_1_D: Optional[torch.Tensor] = None,
        adaln_lora_B_T_3D: Optional[torch.Tensor] = None,
        extra_per_block_pos_emb: Optional[torch.Tensor] = None,
    ) -> torch.Tensor:
        use_fp32 = x_B_T_H_W_D.dtype == torch.float16
        if use_fp32:
            # Cast to float32 for better numerical stability in residual connections. Each module will cast back to float16 by enclosing autocast context.
            x_B_T_H_W_D = x_B_T_H_W_D.float()
@@ -959,6 +959,7 @@ class Block(nn.Module):
        emb_B_T_D: torch.Tensor,
        crossattn_emb: torch.Tensor,
        attn_params: attention.AttentionParams,
        use_fp32: bool = False,
        rope_emb_L_1_1_D: Optional[torch.Tensor] = None,
        adaln_lora_B_T_3D: Optional[torch.Tensor] = None,
        extra_per_block_pos_emb: Optional[torch.Tensor] = None,
@@ -972,6 +973,7 @@ class Block(nn.Module):
                    emb_B_T_D,
                    crossattn_emb,
                    attn_params,
                    use_fp32,
                    rope_emb_L_1_1_D,
                    adaln_lora_B_T_3D,
                    extra_per_block_pos_emb,
@@ -994,6 +996,7 @@ class Block(nn.Module):
                    emb_B_T_D,
                    crossattn_emb,
                    attn_params,
                    use_fp32,
                    rope_emb_L_1_1_D,
                    adaln_lora_B_T_3D,
                    extra_per_block_pos_emb,
@@ -1007,6 +1010,7 @@ class Block(nn.Module):
                    emb_B_T_D,
                    crossattn_emb,
                    attn_params,
                    use_fp32,
                    rope_emb_L_1_1_D,
                    adaln_lora_B_T_3D,
                    extra_per_block_pos_emb,
@@ -1018,6 +1022,7 @@ class Block(nn.Module):
                emb_B_T_D,
                crossattn_emb,
                attn_params,
                use_fp32,
                rope_emb_L_1_1_D,
                adaln_lora_B_T_3D,
                extra_per_block_pos_emb,
@@ -1338,16 +1343,19 @@ class Anima(nn.Module):
        attn_params = attention.AttentionParams.create_attention_params(self.attn_mode, self.split_attn)
        # Determine whether to use float32 for block computations based on input dtype (use float32 for better stability when input is float16)
        use_fp32 = x_B_T_H_W_D.dtype == torch.float16
        for block_idx, block in enumerate(self.blocks):
            if self.blocks_to_swap:
                self.offloader.wait_for_block(block_idx)
-            x_B_T_H_W_D = block(x_B_T_H_W_D, t_embedding_B_T_D, crossattn_emb, attn_params, **block_kwargs)
+            x_B_T_H_W_D = block(x_B_T_H_W_D, t_embedding_B_T_D, crossattn_emb, attn_params, use_fp32, **block_kwargs)
            if self.blocks_to_swap:
                self.offloader.submit_move_blocks(self.blocks, block_idx)
-        x_B_T_H_W_O = self.final_layer(x_B_T_H_W_D, t_embedding_B_T_D, adaln_lora_B_T_3D=adaln_lora_B_T_3D)
+        x_B_T_H_W_O = self.final_layer(x_B_T_H_W_D, t_embedding_B_T_D, adaln_lora_B_T_3D=adaln_lora_B_T_3D, use_fp32=use_fp32)
        x_B_C_Tt_Hp_Wp = self.unpatchify(x_B_T_H_W_O)
        return x_B_C_Tt_Hp_Wp
--- a/library/train_util.py
+++ b/library/train_util.py
@@ -64,7 +64,12 @@ from library import custom_train_functions, sd3_utils
 from library.original_unet import UNet2DConditionModel
 from huggingface_hub import hf_hub_download
 import numpy as np
 import sys
 from PIL import Image
 try:
    from PIL import ImageCms
 except:
    print( "ImageCms not available. Images will not be converted to sRGB. Colours may be handled incorrectly." )
 import imagesize
 import cv2
 import safetensors.torch
@@ -3004,10 +3009,36 @@ def load_arbitrary_dataset(args, tokenizer=None) -> MinimalDataset:
 def load_image(image_path, alpha=False):
    try:
        with Image.open(image_path) as image:
            if getattr(image, "is_animated", False):
                logger.warning( f"{image_path} is animated" )
            # Convert image to sRGB
            if "PIL.ImageCms" in sys.modules:
                icc = image.info.get('icc_profile', '')
                if icc:
                    try:
                        src_profile = ImageCms.ImageCmsProfile( BytesIO(icc) )
                        srgb_profile = ImageCms.createProfile("sRGB")
                        ImageCms.profileToProfile(image, src_profile, srgb_profile, inPlace=True)
                        image.info["icc_profile"] = ImageCms.ImageCmsProfile(srgb_profile).tobytes()
                    except Exception as e:
                        logger.warning( f"Could not convert {image_path} to sRGB: {src_profile.profile.model} {src_profile.profile.profile_description}\n{e}" )
            if alpha:
                if not image.mode == "RGBA":
                    image = image.convert("RGBA")
            else:
                if image.mode == "P":
                    # Palette images with alpha are easier to handle as RGBA.
                    image = image.convert('RGBA')
                if "A" in  image.getbands():
                    # Replace transparency with white background.
                    alpha_layer = image.convert('RGBA').split()[-1]
                    bg = Image.new("RGBA", image.size, (255, 255, 255, 255) )
                    bg.paste( image, mask=alpha_layer )
                    image = bg.convert('RGB')
                if not image.mode == "RGB":
                    image = image.convert("RGB")
            img = np.array(image, np.uint8)
Author	SHA1	Message	Date
Hirmuolio	330ea4acfa	Merge `c19c90d66e` into `51435f1718`	2026-04-05 02:56:20 +00:00
Kohya S.	51435f1718	Merge pull request #2303 from kohya-ss/sd3 fix: improve numerical stability by conditionally using float32 in Anima with fp16 training	2026-04-02 12:40:48 +09:00
Kohya S.	fa53f71ec0	fix: improve numerical stability by conditionally using float32 in Anima (#2302 ) * fix: improve numerical stability by conditionally using float32 in block computations * doc: update README for improvement stability for fp16 training on Anima in version 0.10.3	2026-04-02 12:36:29 +09:00
JM	c19c90d66e	replace errors with warnings	2025-10-05 15:33:41 +03:00
JM	b9ddf0ce37	works without ImageCms	2025-09-28 15:10:09 +03:00
JM	1d4c578ac8	move color profile creation inside try	2025-09-28 14:04:11 +03:00
JM	8fb676cb69	fix reused variable	2025-09-28 13:59:03 +03:00
JM	f73221038e	show filenames in errors	2025-09-27 16:45:21 +03:00
JM	4446d4b58d	better non srgb handling	2025-09-27 16:33:22 +03:00