From a40638f853e72f3de38b11c1ca11038c5bb34f46 Mon Sep 17 00:00:00 2001 From: Bubbliiiing <47347516+bubbliiiing@users.noreply.github.com> Date: Mon, 15 Sep 2025 13:45:54 +0800 Subject: [PATCH] Update lora load in comfyui && Fix bug in qwen image training. (#318) --- ...loading_workflow_subjects_4steps_lora.json | 892 ++++++++++++++++++ ...hunked_loading_workflow_subjects_lora.json | 892 ++++++++++++++++++ scripts/qwenimage/train.py | 4 +- videox_fun/models/qwenimage_transformer2d.py | 172 +++- videox_fun/utils/lora_utils.py | 12 + 5 files changed, 1960 insertions(+), 12 deletions(-) create mode 100644 comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_4steps_lora.json create mode 100644 comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_lora.json diff --git a/comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_4steps_lora.json b/comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_4steps_lora.json new file mode 100644 index 0000000..fdfa5a4 --- /dev/null +++ b/comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_4steps_lora.json @@ -0,0 +1,892 @@ +{ + "id": "71b53971-2ab9-41eb-a405-78c174d97672", + "revision": 0, + "last_node_id": 111, + "last_link_id": 93, + "nodes": [ + { + "id": 78, + "type": "Note", + "pos": [ + 18, + -46 + ], + "size": [ + 210, + 88 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "You can write prompt here\n(你可以在此填写提示词)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 88, + "type": "Note", + "pos": [ + -99, + 197 + ], + "size": [ + 326.1556091308594, + 145.20904541015625 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 94, + "type": "FunTextBox", + "pos": [ + 258, + 178 + ], + "size": [ + 368.5529479980469, + 159.4075927734375 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "prompt", + "type": "STRING_PROMPT", + "slot_index": 0, + "links": [ + 77 + ] + } + ], + "properties": { + "Node name for S&R": "FunTextBox" + }, + "widgets_values": [ + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走" + ] + }, + { + "id": 17, + "type": "VHS_VideoCombine", + "pos": [ + 1179.8734130859375, + -215.58523559570312 + ], + "size": [ + 390.9534912109375, + 966.9860229492188 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "label": "图像", + "name": "images", + "shape": 7, + "type": "IMAGE", + "link": 80 + }, + { + "label": "音频", + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "label": "批次管理", + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "label": "文件名", + "name": "Filenames", + "type": "VHS_FILENAMES", + "slot_index": 0, + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "Fun", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 22, + "save_metadata": true, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "Fun_00154.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 16 + } + } + } + }, + { + "id": 79, + "type": "Note", + "pos": [ + -215.824951171875, + 490.2828369140625 + ], + "size": [ + 210, + 88 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "You can upload ref images here\n(在此上传参考图片)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 104, + "type": "ImageCollectNode", + "pos": [ + 835.30908203125, + 491.93438720703125 + ], + "size": [ + 162.85116577148438, + 46 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "image_1", + "type": "IMAGE", + "link": 83 + }, + { + "name": "image_2", + "shape": 7, + "type": "IMAGE", + "link": 82 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 84 + ] + } + ], + "properties": { + "Node name for S&R": "ImageCollectNode" + }, + "widgets_values": [] + }, + { + "id": 103, + "type": "LoadImage", + "pos": [ + 409.56060791015625, + 457.4063415527344 + ], + "size": [ + 315, + 314.0000305175781 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 82 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "ref_1.png", + "image" + ] + }, + { + "id": 92, + "type": "FunTextBox", + "pos": [ + 254, + -46 + ], + "size": [ + 380.845703125, + 157.68350219726562 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "prompt", + "type": "STRING_PROMPT", + "slot_index": 0, + "links": [ + 76 + ] + } + ], + "title": "Positive Prompt(正向提示词)", + "properties": { + "Node name for S&R": "FunTextBox" + }, + "widgets_values": [ + "海风作曲,浪花打拍。她握着一台亮黄色相机,双臂如翼般舒展流转。双手托举相机,在胸前划出轻柔的圆弧,时而高举过肩,迎向天际的光晕,时而缓缓贴近心口,倾听快门低语般的节奏。紫色长发在风中扬起,拂过肩头,仿佛与镜头共舞。她的目光随取景框游走,每一次微倾与回旋,都让镜头掠过海天相接的边际、摇曳的花丛。最后,呼吸一凝,稳稳对准水中倒影,轻轻按下快门——将风、光与心动,悄然收藏。" + ] + }, + { + "id": 97, + "type": "LoadImage", + "pos": [ + 51.6423454284668, + 459.73138427734375 + ], + "size": [ + 315, + 314.0000305175781 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 83 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "8.png", + "image" + ] + }, + { + "id": 105, + "type": "LoadVaceWanTransformer3DModel", + "pos": [ + 238.0269317626953, + -745.763671875 + ], + "size": [ + 408.1667785644531, + 102 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "transformer", + "type": "TransformerModel", + "links": [ + 85 + ] + }, + { + "name": "model_name", + "type": "STRING", + "links": [ + 90 + ] + } + ], + "properties": { + "Node name for S&R": "LoadVaceWanTransformer3DModel" + }, + "widgets_values": [ + "Wan2.2-VACE-Fun-A14B-LOW_bf16.safetensors", + "bf16" + ] + }, + { + "id": 106, + "type": "LoadWanVAEModel", + "pos": [ + 240.40968322753906, + -583.1104736328125 + ], + "size": [ + 406.1996154785156, + 82 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "vae", + "type": "VAEModel", + "links": [ + 86 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWanVAEModel" + }, + "widgets_values": [ + "Wan2.1_VAE.pth", + "bf16" + ] + }, + { + "id": 107, + "type": "LoadWanTextEncoderModel", + "pos": [ + 241.08558654785156, + -434.9287109375 + ], + "size": [ + 404.7171936035156, + 102 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "text_encoder", + "type": "TextEncoderModel", + "links": [ + 87 + ] + }, + { + "name": "tokenizer", + "type": "Tokenizer", + "links": [ + 88 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWanTextEncoderModel" + }, + "widgets_values": [ + "models_t5_umt5-xxl-enc-bf16.pth", + "bf16" + ] + }, + { + "id": 109, + "type": "LoadVaceWanTransformer3DModel", + "pos": [ + 692.4532470703125, + -742.7451782226562 + ], + "size": [ + 354.0464782714844, + 102 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "transformer", + "type": "TransformerModel", + "links": [ + 89 + ] + }, + { + "name": "model_name", + "type": "STRING", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadVaceWanTransformer3DModel" + }, + "widgets_values": [ + "Wan2.2-VACE-Fun-A14B-HIGH_bf16.safetensors", + "bf16" + ] + }, + { + "id": 110, + "type": "Note", + "pos": [ + -220.1790008544922, + -468.36676025390625 + ], + "size": [ + 427.074951171875, + 143.9142608642578 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "When using the 14B model, you can use sequential_cpu_offload to save GPU memory during generation.\n(在使用14B模型时,可以使用sequential_cpu_offload节省显存,进行生成。)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 108, + "type": "CombineWan2_2VaceFunPipeline", + "pos": [ + 693.27978515625, + -566.5641479492188 + ], + "size": [ + 361.8251953125, + 182 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "transformer", + "type": "TransformerModel", + "link": 85 + }, + { + "name": "vae", + "type": "VAEModel", + "link": 86 + }, + { + "name": "text_encoder", + "type": "TextEncoderModel", + "link": 87 + }, + { + "name": "tokenizer", + "type": "Tokenizer", + "link": 88 + }, + { + "name": "clip_encoder", + "shape": 7, + "type": "ClipEncoderModel", + "link": null + }, + { + "name": "transformer_2", + "shape": 7, + "type": "TransformerModel", + "link": 89 + }, + { + "name": "model_name", + "type": "STRING", + "widget": { + "name": "model_name" + }, + "link": 90 + } + ], + "outputs": [ + { + "name": "funmodels", + "type": "FunModels", + "links": [ + 92 + ] + } + ], + "properties": { + "Node name for S&R": "CombineWan2_2VaceFunPipeline" + }, + "widgets_values": [ + "", + "sequential_cpu_offload" + ] + }, + { + "id": 111, + "type": "LoadWan2_2FunLora", + "pos": [ + 1122.6082763671875, + -506.19036865234375 + ], + "size": [ + 601.3900756835938, + 130 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "funmodels", + "type": "FunModels", + "link": 92 + } + ], + "outputs": [ + { + "name": "funmodels", + "type": "FunModels", + "links": [ + 93 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWan2_2FunLora" + }, + "widgets_values": [ + "Wan2.2-T2V-A14B-4steps-lora-rank64-Seko-V1.1_low_noise_model.safetensors", + "Wan2.2-T2V-A14B-4steps-lora-rank64-Seko-V1.1_high_noise_model.safetensors", + 1, + false + ] + }, + { + "id": 101, + "type": "Wan2_2VaceFunSampler", + "pos": [ + 842.1512451171875, + -215.00350952148438 + ], + "size": [ + 280.724609375, + 534 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "funmodels", + "type": "FunModels", + "link": 93 + }, + { + "name": "prompt", + "type": "STRING_PROMPT", + "link": 76 + }, + { + "name": "negative_prompt", + "type": "STRING_PROMPT", + "link": 77 + }, + { + "name": "control_video", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "start_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "end_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "subject_ref_images", + "shape": 7, + "type": "IMAGE", + "link": 84 + }, + { + "name": "riflex_k", + "shape": 7, + "type": "RIFLEXT_ARGS", + "link": null + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 80 + ] + } + ], + "properties": { + "Node name for S&R": "Wan2_2VaceFunSampler" + }, + "widgets_values": [ + 81, + 640, + 580941324226511, + "randomize", + 4, + 1.0000000000000002, + "Flow_Unipc", + 12, + 0.875, + 0.1, + false, + 3, + true, + 0, + true + ] + } + ], + "links": [ + [ + 76, + 92, + 0, + 101, + 1, + "STRING_PROMPT" + ], + [ + 77, + 94, + 0, + 101, + 2, + "STRING_PROMPT" + ], + [ + 80, + 101, + 0, + 17, + 0, + "IMAGE" + ], + [ + 82, + 103, + 0, + 104, + 1, + "IMAGE" + ], + [ + 83, + 97, + 0, + 104, + 0, + "IMAGE" + ], + [ + 84, + 104, + 0, + 101, + 6, + "IMAGE" + ], + [ + 85, + 105, + 0, + 108, + 0, + "TransformerModel" + ], + [ + 86, + 106, + 0, + 108, + 1, + "VAEModel" + ], + [ + 87, + 107, + 0, + 108, + 2, + "TextEncoderModel" + ], + [ + 88, + 107, + 1, + 108, + 3, + "Tokenizer" + ], + [ + 89, + 109, + 0, + 108, + 5, + "TransformerModel" + ], + [ + 90, + 105, + 1, + 108, + 6, + "STRING" + ], + [ + 92, + 108, + 0, + 111, + 0, + "FunModels" + ], + [ + 93, + 111, + 0, + 101, + 0, + "FunModels" + ] + ], + "groups": [ + { + "id": 1, + "title": "Upload Your Ref Images", + "bounding": [ + -0.4573218524456024, + 378.5783386230469, + 1071.616943359375, + 451.41473388671875 + ], + "color": "#a1309b", + "font_size": 24, + "flags": {} + }, + { + "id": 3, + "title": "Prompts", + "bounding": [ + 218, + -127, + 450, + 483 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 4, + "title": "Load Model", + "bounding": [ + 220.02639770507812, + -821.3883056640625, + 863.243896484375, + 511.2655029296875 + ], + "color": "#b06634", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "ds": { + "scale": 1.1, + "offset": [ + -67.03623665930887, + 397.7621525676092 + ] + }, + "frontendVersion": "1.21.3", + "workspace_info": { + "id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea" + }, + "node_versions": { + "CogVideoX-Fun": "abeac38889da94b1f215b456a8180198a24d5032", + "ComfyUI-VideoHelperSuite": "70faa9bcef65932ab72e7404d6373fb300013a2e", + "comfy-core": "0.3.57" + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_lora.json b/comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_lora.json new file mode 100644 index 0000000..0bb0453 --- /dev/null +++ b/comfyui/wan2_2_vace_fun/v1/wan2.2_vace_fun_chunked_loading_workflow_subjects_lora.json @@ -0,0 +1,892 @@ +{ + "id": "71b53971-2ab9-41eb-a405-78c174d97672", + "revision": 0, + "last_node_id": 111, + "last_link_id": 93, + "nodes": [ + { + "id": 78, + "type": "Note", + "pos": [ + 18, + -46 + ], + "size": [ + 210, + 88 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "You can write prompt here\n(你可以在此填写提示词)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 88, + "type": "Note", + "pos": [ + -99, + 197 + ], + "size": [ + 326.1556091308594, + 145.20904541015625 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 94, + "type": "FunTextBox", + "pos": [ + 258, + 178 + ], + "size": [ + 368.5529479980469, + 159.4075927734375 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "prompt", + "type": "STRING_PROMPT", + "slot_index": 0, + "links": [ + 77 + ] + } + ], + "properties": { + "Node name for S&R": "FunTextBox" + }, + "widgets_values": [ + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走" + ] + }, + { + "id": 17, + "type": "VHS_VideoCombine", + "pos": [ + 1179.8734130859375, + -215.58523559570312 + ], + "size": [ + 390.9534912109375, + 966.9860229492188 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "label": "图像", + "name": "images", + "shape": 7, + "type": "IMAGE", + "link": 80 + }, + { + "label": "音频", + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "label": "批次管理", + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "label": "文件名", + "name": "Filenames", + "type": "VHS_FILENAMES", + "slot_index": 0, + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "Fun", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 22, + "save_metadata": true, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "Fun_00154.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 16 + } + } + } + }, + { + "id": 79, + "type": "Note", + "pos": [ + -215.824951171875, + 490.2828369140625 + ], + "size": [ + 210, + 88 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "You can upload ref images here\n(在此上传参考图片)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 104, + "type": "ImageCollectNode", + "pos": [ + 835.30908203125, + 491.93438720703125 + ], + "size": [ + 162.85116577148438, + 46 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "image_1", + "type": "IMAGE", + "link": 83 + }, + { + "name": "image_2", + "shape": 7, + "type": "IMAGE", + "link": 82 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 84 + ] + } + ], + "properties": { + "Node name for S&R": "ImageCollectNode" + }, + "widgets_values": [] + }, + { + "id": 103, + "type": "LoadImage", + "pos": [ + 409.56060791015625, + 457.4063415527344 + ], + "size": [ + 315, + 314.0000305175781 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 82 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "ref_1.png", + "image" + ] + }, + { + "id": 92, + "type": "FunTextBox", + "pos": [ + 254, + -46 + ], + "size": [ + 380.845703125, + 157.68350219726562 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "prompt", + "type": "STRING_PROMPT", + "slot_index": 0, + "links": [ + 76 + ] + } + ], + "title": "Positive Prompt(正向提示词)", + "properties": { + "Node name for S&R": "FunTextBox" + }, + "widgets_values": [ + "海风作曲,浪花打拍。她握着一台亮黄色相机,双臂如翼般舒展流转。双手托举相机,在胸前划出轻柔的圆弧,时而高举过肩,迎向天际的光晕,时而缓缓贴近心口,倾听快门低语般的节奏。紫色长发在风中扬起,拂过肩头,仿佛与镜头共舞。她的目光随取景框游走,每一次微倾与回旋,都让镜头掠过海天相接的边际、摇曳的花丛。最后,呼吸一凝,稳稳对准水中倒影,轻轻按下快门——将风、光与心动,悄然收藏。" + ] + }, + { + "id": 97, + "type": "LoadImage", + "pos": [ + 51.6423454284668, + 459.73138427734375 + ], + "size": [ + 315, + 314.0000305175781 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 83 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "8.png", + "image" + ] + }, + { + "id": 105, + "type": "LoadVaceWanTransformer3DModel", + "pos": [ + 238.0269317626953, + -745.763671875 + ], + "size": [ + 408.1667785644531, + 102 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "transformer", + "type": "TransformerModel", + "links": [ + 85 + ] + }, + { + "name": "model_name", + "type": "STRING", + "links": [ + 90 + ] + } + ], + "properties": { + "Node name for S&R": "LoadVaceWanTransformer3DModel" + }, + "widgets_values": [ + "Wan2.2-VACE-Fun-A14B-LOW_bf16.safetensors", + "bf16" + ] + }, + { + "id": 106, + "type": "LoadWanVAEModel", + "pos": [ + 240.40968322753906, + -583.1104736328125 + ], + "size": [ + 406.1996154785156, + 82 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "vae", + "type": "VAEModel", + "links": [ + 86 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWanVAEModel" + }, + "widgets_values": [ + "Wan2.1_VAE.pth", + "bf16" + ] + }, + { + "id": 107, + "type": "LoadWanTextEncoderModel", + "pos": [ + 241.08558654785156, + -434.9287109375 + ], + "size": [ + 404.7171936035156, + 102 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "text_encoder", + "type": "TextEncoderModel", + "links": [ + 87 + ] + }, + { + "name": "tokenizer", + "type": "Tokenizer", + "links": [ + 88 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWanTextEncoderModel" + }, + "widgets_values": [ + "models_t5_umt5-xxl-enc-bf16.pth", + "bf16" + ] + }, + { + "id": 109, + "type": "LoadVaceWanTransformer3DModel", + "pos": [ + 692.4532470703125, + -742.7451782226562 + ], + "size": [ + 354.0464782714844, + 102 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "transformer", + "type": "TransformerModel", + "links": [ + 89 + ] + }, + { + "name": "model_name", + "type": "STRING", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadVaceWanTransformer3DModel" + }, + "widgets_values": [ + "Wan2.2-VACE-Fun-A14B-HIGH_bf16.safetensors", + "bf16" + ] + }, + { + "id": 110, + "type": "Note", + "pos": [ + -220.1790008544922, + -468.36676025390625 + ], + "size": [ + 427.074951171875, + 143.9142608642578 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "When using the 14B model, you can use sequential_cpu_offload to save GPU memory during generation.\n(在使用14B模型时,可以使用sequential_cpu_offload节省显存,进行生成。)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 108, + "type": "CombineWan2_2VaceFunPipeline", + "pos": [ + 693.27978515625, + -566.5641479492188 + ], + "size": [ + 361.8251953125, + 182 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "transformer", + "type": "TransformerModel", + "link": 85 + }, + { + "name": "vae", + "type": "VAEModel", + "link": 86 + }, + { + "name": "text_encoder", + "type": "TextEncoderModel", + "link": 87 + }, + { + "name": "tokenizer", + "type": "Tokenizer", + "link": 88 + }, + { + "name": "clip_encoder", + "shape": 7, + "type": "ClipEncoderModel", + "link": null + }, + { + "name": "transformer_2", + "shape": 7, + "type": "TransformerModel", + "link": 89 + }, + { + "name": "model_name", + "type": "STRING", + "widget": { + "name": "model_name" + }, + "link": 90 + } + ], + "outputs": [ + { + "name": "funmodels", + "type": "FunModels", + "links": [ + 92 + ] + } + ], + "properties": { + "Node name for S&R": "CombineWan2_2VaceFunPipeline" + }, + "widgets_values": [ + "", + "sequential_cpu_offload" + ] + }, + { + "id": 111, + "type": "LoadWan2_2FunLora", + "pos": [ + 1122.6082763671875, + -506.19036865234375 + ], + "size": [ + 601.3900756835938, + 130 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "funmodels", + "type": "FunModels", + "link": 92 + } + ], + "outputs": [ + { + "name": "funmodels", + "type": "FunModels", + "links": [ + 93 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWan2_2FunLora" + }, + "widgets_values": [ + "Wan2.2-T2V-A14B-4steps-lora-rank64-Seko-V1.1_low_noise_model.safetensors", + "Wan2.2-T2V-A14B-4steps-lora-rank64-Seko-V1.1_high_noise_model.safetensors", + 1, + false + ] + }, + { + "id": 101, + "type": "Wan2_2VaceFunSampler", + "pos": [ + 842.1512451171875, + -215.00350952148438 + ], + "size": [ + 280.724609375, + 534 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "funmodels", + "type": "FunModels", + "link": 93 + }, + { + "name": "prompt", + "type": "STRING_PROMPT", + "link": 76 + }, + { + "name": "negative_prompt", + "type": "STRING_PROMPT", + "link": 77 + }, + { + "name": "control_video", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "start_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "end_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "subject_ref_images", + "shape": 7, + "type": "IMAGE", + "link": 84 + }, + { + "name": "riflex_k", + "shape": 7, + "type": "RIFLEXT_ARGS", + "link": null + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 80 + ] + } + ], + "properties": { + "Node name for S&R": "Wan2_2VaceFunSampler" + }, + "widgets_values": [ + 81, + 640, + 580941324226511, + "randomize", + 40, + 1.0000000000000002, + "Flow", + 12, + 0.875, + 0.1, + true, + 3, + true, + 0.25000000000000006, + true + ] + } + ], + "links": [ + [ + 76, + 92, + 0, + 101, + 1, + "STRING_PROMPT" + ], + [ + 77, + 94, + 0, + 101, + 2, + "STRING_PROMPT" + ], + [ + 80, + 101, + 0, + 17, + 0, + "IMAGE" + ], + [ + 82, + 103, + 0, + 104, + 1, + "IMAGE" + ], + [ + 83, + 97, + 0, + 104, + 0, + "IMAGE" + ], + [ + 84, + 104, + 0, + 101, + 6, + "IMAGE" + ], + [ + 85, + 105, + 0, + 108, + 0, + "TransformerModel" + ], + [ + 86, + 106, + 0, + 108, + 1, + "VAEModel" + ], + [ + 87, + 107, + 0, + 108, + 2, + "TextEncoderModel" + ], + [ + 88, + 107, + 1, + 108, + 3, + "Tokenizer" + ], + [ + 89, + 109, + 0, + 108, + 5, + "TransformerModel" + ], + [ + 90, + 105, + 1, + 108, + 6, + "STRING" + ], + [ + 92, + 108, + 0, + 111, + 0, + "FunModels" + ], + [ + 93, + 111, + 0, + 101, + 0, + "FunModels" + ] + ], + "groups": [ + { + "id": 1, + "title": "Upload Your Ref Images", + "bounding": [ + -0.4573218524456024, + 378.5783386230469, + 1071.616943359375, + 451.41473388671875 + ], + "color": "#a1309b", + "font_size": 24, + "flags": {} + }, + { + "id": 3, + "title": "Prompts", + "bounding": [ + 218, + -127, + 450, + 483 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 4, + "title": "Load Model", + "bounding": [ + 220.02639770507812, + -821.3883056640625, + 863.243896484375, + 511.2655029296875 + ], + "color": "#b06634", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "ds": { + "scale": 0.9090909090909095, + "offset": [ + -50.780612464991464, + 485.5421122795407 + ] + }, + "frontendVersion": "1.21.3", + "workspace_info": { + "id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea" + }, + "node_versions": { + "CogVideoX-Fun": "abeac38889da94b1f215b456a8180198a24d5032", + "ComfyUI-VideoHelperSuite": "70faa9bcef65932ab72e7404d6373fb300013a2e", + "comfy-core": "0.3.57" + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/scripts/qwenimage/train.py b/scripts/qwenimage/train.py index 4db9d17..cf1e9e1 100644 --- a/scripts/qwenimage/train.py +++ b/scripts/qwenimage/train.py @@ -135,7 +135,7 @@ check_min_version("0.18.0.dev0") logger = get_logger(__name__, log_level="INFO") -def log_validation(vae, text_encoder, tokenizer, transformer3d, args, config, accelerator, weight_dtype, global_step): +def log_validation(vae, text_encoder, tokenizer, transformer3d, args, accelerator, weight_dtype, global_step): try: logger.info("Running validation... ") @@ -1566,7 +1566,6 @@ def main(): tokenizer, transformer3d, args, - config, accelerator, weight_dtype, global_step, @@ -1593,7 +1592,6 @@ def main(): tokenizer, transformer3d, args, - config, accelerator, weight_dtype, global_step, diff --git a/videox_fun/models/qwenimage_transformer2d.py b/videox_fun/models/qwenimage_transformer2d.py index 4dedd4f..ec5accc 100644 --- a/videox_fun/models/qwenimage_transformer2d.py +++ b/videox_fun/models/qwenimage_transformer2d.py @@ -558,6 +558,14 @@ class QwenImageTransformer2DModel(ModelMixin, ConfigMixin, PeftAdapterMixin, Fro self.sp_world_size = 1 self.sp_world_rank = 0 + def _set_gradient_checkpointing(self, *args, **kwargs): + if "value" in kwargs: + self.gradient_checkpointing = kwargs["value"] + elif "enable" in kwargs: + self.gradient_checkpointing = kwargs["enable"] + else: + raise ValueError("Invalid set gradient checkpointing") + def enable_multi_gpus_inference(self,): self.sp_world_size = get_sequence_parallel_world_size() self.sp_world_rank = get_sequence_parallel_rank() @@ -703,14 +711,21 @@ class QwenImageTransformer2DModel(ModelMixin, ConfigMixin, PeftAdapterMixin, Fro for index_block, block in enumerate(self.transformer_blocks): if torch.is_grad_enabled() and self.gradient_checkpointing: - encoder_hidden_states, hidden_states = self._gradient_checkpointing_func( - block, - hidden_states, - encoder_hidden_states, - encoder_hidden_states_mask, - temb, - image_rotary_emb, - ) + def create_custom_forward(module): + def custom_forward(*inputs): + return module(*inputs) + + return custom_forward + ckpt_kwargs: Dict[str, Any] = {"use_reentrant": False} if is_torch_version(">=", "1.11.0") else {} + encoder_hidden_states, hidden_states = torch.utils.checkpoint.checkpoint( + create_custom_forward(block), + hidden_states, + encoder_hidden_states, + encoder_hidden_states_mask, + temb, + image_rotary_emb, + **ckpt_kwargs, + ) else: encoder_hidden_states, hidden_states = block( @@ -736,4 +751,143 @@ class QwenImageTransformer2DModel(ModelMixin, ConfigMixin, PeftAdapterMixin, Fro if not return_dict: return (output,) - return Transformer2DModelOutput(sample=output) \ No newline at end of file + return Transformer2DModelOutput(sample=output) + + @classmethod + def from_pretrained( + cls, pretrained_model_path, subfolder=None, transformer_additional_kwargs={}, + low_cpu_mem_usage=False, torch_dtype=torch.bfloat16 + ): + if subfolder is not None: + pretrained_model_path = os.path.join(pretrained_model_path, subfolder) + print(f"loaded 3D transformer's pretrained weights from {pretrained_model_path} ...") + + config_file = os.path.join(pretrained_model_path, 'config.json') + if not os.path.isfile(config_file): + raise RuntimeError(f"{config_file} does not exist") + with open(config_file, "r") as f: + config = json.load(f) + + from diffusers.utils import WEIGHTS_NAME + model_file = os.path.join(pretrained_model_path, WEIGHTS_NAME) + model_file_safetensors = model_file.replace(".bin", ".safetensors") + + if "dict_mapping" in transformer_additional_kwargs.keys(): + for key in transformer_additional_kwargs["dict_mapping"]: + transformer_additional_kwargs[transformer_additional_kwargs["dict_mapping"][key]] = config[key] + + if low_cpu_mem_usage: + try: + import re + + from diffusers import __version__ as diffusers_version + if diffusers_version >= "0.33.0": + from diffusers.models.model_loading_utils import \ + load_model_dict_into_meta + else: + from diffusers.models.modeling_utils import \ + load_model_dict_into_meta + from diffusers.utils import is_accelerate_available + if is_accelerate_available(): + import accelerate + + # Instantiate model with empty weights + with accelerate.init_empty_weights(): + model = cls.from_config(config, **transformer_additional_kwargs) + + param_device = "cpu" + if os.path.exists(model_file): + state_dict = torch.load(model_file, map_location="cpu") + elif os.path.exists(model_file_safetensors): + from safetensors.torch import load_file, safe_open + state_dict = load_file(model_file_safetensors) + else: + from safetensors.torch import load_file, safe_open + model_files_safetensors = glob.glob(os.path.join(pretrained_model_path, "*.safetensors")) + state_dict = {} + print(model_files_safetensors) + for _model_file_safetensors in model_files_safetensors: + _state_dict = load_file(_model_file_safetensors) + for key in _state_dict: + state_dict[key] = _state_dict[key] + + if diffusers_version >= "0.33.0": + # Diffusers has refactored `load_model_dict_into_meta` since version 0.33.0 in this commit: + # https://github.com/huggingface/diffusers/commit/f5929e03060d56063ff34b25a8308833bec7c785. + load_model_dict_into_meta( + model, + state_dict, + dtype=torch_dtype, + model_name_or_path=pretrained_model_path, + ) + else: + model._convert_deprecated_attention_blocks(state_dict) + # move the params from meta device to cpu + missing_keys = set(model.state_dict().keys()) - set(state_dict.keys()) + if len(missing_keys) > 0: + raise ValueError( + f"Cannot load {cls} from {pretrained_model_path} because the following keys are" + f" missing: \n {', '.join(missing_keys)}. \n Please make sure to pass" + " `low_cpu_mem_usage=False` and `device_map=None` if you want to randomly initialize" + " those weights or else make sure your checkpoint file is correct." + ) + + unexpected_keys = load_model_dict_into_meta( + model, + state_dict, + device=param_device, + dtype=torch_dtype, + model_name_or_path=pretrained_model_path, + ) + + if cls._keys_to_ignore_on_load_unexpected is not None: + for pat in cls._keys_to_ignore_on_load_unexpected: + unexpected_keys = [k for k in unexpected_keys if re.search(pat, k) is None] + + if len(unexpected_keys) > 0: + print( + f"Some weights of the model checkpoint were not used when initializing {cls.__name__}: \n {[', '.join(unexpected_keys)]}" + ) + + return model + except Exception as e: + print( + f"The low_cpu_mem_usage mode is not work because {e}. Use low_cpu_mem_usage=False instead." + ) + + model = cls.from_config(config, **transformer_additional_kwargs) + if os.path.exists(model_file): + state_dict = torch.load(model_file, map_location="cpu") + elif os.path.exists(model_file_safetensors): + from safetensors.torch import load_file, safe_open + state_dict = load_file(model_file_safetensors) + else: + from safetensors.torch import load_file, safe_open + model_files_safetensors = glob.glob(os.path.join(pretrained_model_path, "*.safetensors")) + state_dict = {} + for _model_file_safetensors in model_files_safetensors: + _state_dict = load_file(_model_file_safetensors) + for key in _state_dict: + state_dict[key] = _state_dict[key] + + tmp_state_dict = {} + for key in state_dict: + if key in model.state_dict().keys() and model.state_dict()[key].size() == state_dict[key].size(): + tmp_state_dict[key] = state_dict[key] + else: + print(key, "Size don't match, skip") + + state_dict = tmp_state_dict + + m, u = model.load_state_dict(state_dict, strict=False) + print(f"### missing keys: {len(m)}; \n### unexpected keys: {len(u)};") + print(m) + + params = [p.numel() if "." in n else 0 for n, p in model.named_parameters()] + print(f"### All Parameters: {sum(params) / 1e6} M") + + params = [p.numel() if "attn1." in n else 0 for n, p in model.named_parameters()] + print(f"### attn1 Parameters: {sum(params) / 1e6} M") + + model = model.to(torch_dtype) + return model \ No newline at end of file diff --git a/videox_fun/utils/lora_utils.py b/videox_fun/utils/lora_utils.py index 59d7d94..b076625 100755 --- a/videox_fun/utils/lora_utils.py +++ b/videox_fun/utils/lora_utils.py @@ -377,6 +377,12 @@ def merge_lora(pipeline, lora_path, multiplier, device='cpu', dtype=torch.float3 state_dict = state_dict updates = defaultdict(dict) for key, value in state_dict.items(): + if "diffusion_model" in key: + key = key.replace("diffusion_model.", "lora_unet__") + key = key.replace("blocks.", "blocks_") + key = key.replace(".self_attn.", "_self_attn_") + key = key.replace(".cross_attn.", "_cross_attn_") + key = key.replace(".ffn.", "_ffn_") layer, elem = key.split('.', 1) updates[layer][elem] = value @@ -484,6 +490,12 @@ def unmerge_lora(pipeline, lora_path, multiplier=1, device="cpu", dtype=torch.fl updates = defaultdict(dict) for key, value in state_dict.items(): + if "diffusion_model" in key: + key = key.replace("diffusion_model.", "lora_unet__") + key = key.replace("blocks.", "blocks_") + key = key.replace(".self_attn.", "_self_attn_") + key = key.replace(".cross_attn.", "_cross_attn_") + key = key.replace(".ffn.", "_ffn_") layer, elem = key.split('.', 1) updates[layer][elem] = value