diff --git a/nodes.py b/nodes.py index e645ef5..d114290 100644 --- a/nodes.py +++ b/nodes.py @@ -396,7 +396,7 @@ class InitFluxLoRATraining: "t5xxl_max_token_length": 512, "alpha_mask": dataset["alpha_mask"], "network_train_unet_only": True if train_clip_l == 'disabled' else False, - "fp8_base_unet": True if train_clip_l!='use_fp8' else False, + "fp8_base_unet": True if train_clip_l=='use_gradient_dtype' else False, } attention_settings = { "sdpa": {"mem_eff_attn": True, "xformers": False, "spda": True}, diff --git a/train_network.py b/train_network.py index 2292e17..754b19b 100644 --- a/train_network.py +++ b/train_network.py @@ -539,7 +539,7 @@ class NetworkTrainer: accelerator.print("enable fp8 training for U-Net.") unet_weight_dtype = torch.float8_e4m3fn - if not args.fp8_base_unet: + if not args.fp8_base_unet and not args.network_train_unet_only: accelerator.print("enable fp8 training for Text Encoder.") te_weight_dtype = weight_dtype if args.fp8_base_unet else torch.float8_e4m3fn