From 022ddfd619619ef918220f0ea10749ef650a5943 Mon Sep 17 00:00:00 2001 From: gaclove Date: Thu, 24 Jul 2025 18:34:23 +0800 Subject: [PATCH] feat: enhance ModularConfigManager and LightX2VInferenceConfig with new video duration and adaptive resize options; update inference handling to support audio output --- bridge.py | 6 + examples/wan_i2v.json | 110 ++++--- examples/wan_i2v_with_audio.json | 388 ++++++++++++++++++++++++ examples/wan_i2v_with_distill_lora.json | 244 +++++++++++---- examples/wan_t2v_with_distill_lora.json | 284 +++++++++++------ lightx2v | 2 +- nodes.py | 100 +++--- 7 files changed, 899 insertions(+), 235 deletions(-) create mode 100644 examples/wan_i2v_with_audio.json diff --git a/bridge.py b/bridge.py index 3618e94..ec1a424 100644 --- a/bridge.py +++ b/bridge.py @@ -324,6 +324,12 @@ class ModularConfigManager: if "fps" in config: updates["fps"] = config["fps"] + if "video_duration" in config: + updates["video_duration"] = config["video_duration"] + + if "adaptive_resize" in config: + updates["adaptive_resize"] = config["adaptive_resize"] + if "denoising_step_list" in config: updates["denoising_step_list"] = config["denoising_step_list"] diff --git a/examples/wan_i2v.json b/examples/wan_i2v.json index 3eb3665..3ab15ae 100644 --- a/examples/wan_i2v.json +++ b/examples/wan_i2v.json @@ -107,49 +107,6 @@ }, "widgets_values": [] }, - { - "id": 105, - "type": "LightX2VInferenceConfig", - "pos": [ - 1073.62646484375, - -580.9854125976562 - ], - "size": [ - 270, - 346 - ], - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "inference_config", - "type": "INFERENCE_CONFIG", - "links": [ - 76 - ] - } - ], - "properties": { - "Node name for S&R": "LightX2VInferenceConfig" - }, - "widgets_values": [ - "wan2.1", - "Wan2.1-I2V-14B-480P", - "i2v", - 40, - 1822974886, - "fixed", - 5, - 3, - 480, - 480, - 81, - 16, - "" - ] - }, { "id": 110, "type": "LightX2VModularInference", @@ -190,6 +147,11 @@ "links": [ 81 ] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null } ], "properties": { @@ -209,7 +171,7 @@ ], "size": [ 220.5830078125, - 524.5830078125 + 334 ], "flags": {}, "order": 5, @@ -257,6 +219,7 @@ "pix_fmt": "yuv420p", "crf": 19, "save_metadata": true, + "trim_to_audio": false, "pingpong": false, "save_output": true, "videopreview": { @@ -277,8 +240,8 @@ "id": 111, "type": "easy showAnything", "pos": [ - 1773.56689453125, - -706.2327270507812 + 1766.246337890625, + -545.1817016601562 ], "size": [ 624.5454711914062, @@ -310,6 +273,49 @@ "widgets_values": [ "{\"model_cls\": \"wan2.1\", \"model_path\": \"/mnt/aigc/users/lijiaqi2/ComfyUI/models/lightx2v/Wan2.1-I2V-14B-480P\", \"task\": \"i2v\", \"mode\": \"infer\", \"infer_steps\": 40, \"seed\": 1822974886, \"sample_guide_scale\": 5.0, \"sample_shift\": 3, \"enable_cfg\": true, \"prompt\": \"\", \"negative_prompt\": \"\", \"target_height\": 480, \"target_width\": 480, \"target_video_length\": 81, \"fps\": 16, \"vae_stride\": [4, 8, 8], \"patch_size\": [1, 2, 2], \"feature_caching\": \"NoCaching\", \"teacache_thresh\": 0.26, \"coefficients\": null, \"use_ret_steps\": false, \"dit_quant_scheme\": \"bf16\", \"t5_quant_scheme\": \"bf16\", \"clip_quant_scheme\": \"fp16\", \"quant_op\": \"vllm\", \"precision_mode\": \"fp32\", \"dit_quantized_ckpt\": null, \"t5_quantized_ckpt\": null, \"clip_quantized_ckpt\": null, \"mm_config\": {\"mm_type\": \"Default\"}, \"rotary_chunk\": false, \"rotary_chunk_size\": 100, \"clean_cuda_cache\": false, \"torch_compile\": false, \"attention_type\": \"flash_attn3\", \"self_attn_1_type\": \"flash_attn3\", \"cross_attn_1_type\": \"flash_attn3\", \"cross_attn_2_type\": \"flash_attn3\", \"cpu_offload\": false, \"offload_granularity\": \"phase\", \"offload_ratio\": 1.0, \"t5_cpu_offload\": false, \"t5_offload_granularity\": \"model\", \"lazy_load\": false, \"unload_modules\": false, \"use_tiny_vae\": false, \"tiny_vae\": false, \"tiny_vae_path\": null, \"use_tiling_vae\": false, \"lora_path\": null, \"strength_model\": 1.0, \"do_mm_calib\": false, \"parallel_attn_type\": null, \"parallel_vae\": false, \"max_area\": false, \"use_prompt_enhancer\": false, \"text_len\": 512, \"_class_name\": \"WanModel\", \"_diffusers_version\": \"0.30.0\", \"dim\": 5120, \"eps\": 1e-06, \"ffn_dim\": 13824, \"freq_dim\": 256, \"in_dim\": 36, \"model_type\": \"i2v\", \"num_heads\": 40, \"num_layers\": 40, \"out_dim\": 16}" ] + }, + { + "id": 105, + "type": "LightX2VInferenceConfig", + "pos": [ + 1073.62646484375, + -580.9854125976562 + ], + "size": [ + 270, + 346 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "inference_config", + "type": "INFERENCE_CONFIG", + "links": [ + 76 + ] + } + ], + "properties": { + "Node name for S&R": "LightX2VInferenceConfig" + }, + "widgets_values": [ + "wan2.1", + "Wan2.1-I2V-14B-480P", + "i2v", + 40, + 1822974886, + "fixed", + 5, + 3, + 480, + 480, + 5, + "", + "" + ] } ], "links": [ @@ -360,11 +366,15 @@ "ds": { "scale": 1, "offset": [ - -697.2225740954825, - 708.7013290998841 + -584.0960104591186, + 687.7305563726111 ] }, - "frontendVersion": "1.19.9" + "frontendVersion": "1.23.4", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true }, "version": 0.4 -} +} \ No newline at end of file diff --git a/examples/wan_i2v_with_audio.json b/examples/wan_i2v_with_audio.json new file mode 100644 index 0000000..aef28b3 --- /dev/null +++ b/examples/wan_i2v_with_audio.json @@ -0,0 +1,388 @@ +{ + "id": "8e881096-4f7b-4633-b34b-6b9c0f8aa093", + "revision": 0, + "last_node_id": 118, + "last_link_id": 89, + "nodes": [ + { + "id": 109, + "type": "LightX2VConfigCombiner", + "pos": [1090.0107421875, -940.6998901367188], + "size": [239.138671875, 126], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "inference_config", + "type": "INFERENCE_CONFIG", + "link": 76 + }, + { + "name": "teacache_config", + "shape": 7, + "type": "TEACACHE_CONFIG", + "link": null + }, + { + "name": "quantization_config", + "shape": 7, + "type": "QUANT_CONFIG", + "link": null + }, + { + "name": "memory_config", + "shape": 7, + "type": "MEMORY_CONFIG", + "link": null + }, + { + "name": "vae_config", + "shape": 7, + "type": "VAE_CONFIG", + "link": null + }, + { + "name": "lora_chain", + "shape": 7, + "type": "LORA_CHAIN", + "link": null + } + ], + "outputs": [ + { + "name": "combined_config", + "type": "COMBINED_CONFIG", + "links": [78, 80] + } + ], + "properties": { + "Node name for S&R": "LightX2VConfigCombiner" + }, + "widgets_values": [] + }, + { + "id": 110, + "type": "LightX2VModularInference", + "pos": [1451.4600830078125, -804.5155029296875], + "size": [400, 200], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "combined_config", + "type": "COMBINED_CONFIG", + "link": 78 + }, + { + "name": "image", + "shape": 7, + "type": "IMAGE", + "link": 79 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 85 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [86] + }, + { + "name": "audio", + "type": "AUDIO", + "links": [84] + } + ], + "properties": { + "Node name for S&R": "LightX2VModularInference" + }, + "widgets_values": [ + "The video features a old lady is saying something and knitting a sweater.", + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走" + ] + }, + { + "id": 105, + "type": "LightX2VInferenceConfig", + "pos": [775.7962646484375, -936.5638427734375], + "size": [270, 346], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "inference_config", + "type": "INFERENCE_CONFIG", + "links": [76] + } + ], + "properties": { + "Node name for S&R": "LightX2VInferenceConfig" + }, + "widgets_values": [ + "wan2.1_audio", + "Wan2.1-R2V721-Audio-14B-720P", + "i2v", + 4, + 407287721, + "randomize", + 1, + 5, + 720, + 1280, + 5, + "", + "" + ] + }, + { + "id": 117, + "type": "RIFEInterpolation", + "pos": [1999.54443359375, -796.881591796875], + "size": [270, 130], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 86 + }, + { + "name": "target_fps", + "type": "FLOAT", + "widget": { + "name": "target_fps" + }, + "link": 88 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [87] + } + ], + "properties": { + "Node name for S&R": "RIFEInterpolation" + }, + "widgets_values": [16, 24, 1, "flownet.pkl"] + }, + { + "id": 108, + "type": "VHS_VideoCombine", + "pos": [2351.4580078125, -794.34375], + "size": [220.5830078125, 496.25701904296875], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 87 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 84 + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + }, + { + "name": "frame_rate", + "type": "FLOAT", + "widget": { + "name": "frame_rate" + }, + "link": 89 + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 24, + "loop_count": 0, + "filename_prefix": "AnimateDiff", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "AnimateDiff_00013-audio.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 24, + "workflow": "AnimateDiff_00013.png", + "fullpath": "/mnt/aigc/users/gaopeng1/ComfyUI/output/AnimateDiff_00013-audio.mp4" + }, + "muted": false + } + } + }, + { + "id": 107, + "type": "LoadImage", + "pos": [1041.3621826171875, -540.7197265625], + "size": [274.080078125, 314.0000305175781], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [79] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": ["15.png", "image"] + }, + { + "id": 116, + "type": "LoadAudio", + "pos": [1034.8560791015625, -166.87115478515625], + "size": [270, 136], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [85] + } + ], + "properties": { + "Node name for S&R": "LoadAudio" + }, + "widgets_values": ["15.wav", null, null] + }, + { + "id": 111, + "type": "easy showAnything", + "pos": [1471.281982421875, -908.9144897460938], + "size": [624.5454711914062, 358.7272644042969], + "flags": { + "collapsed": true + }, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "anything", + "shape": 7, + "type": "*", + "link": 80 + } + ], + "outputs": [ + { + "name": "output", + "type": "*", + "links": null + } + ], + "title": "config", + "properties": { + "Node name for S&R": "easy showAnything" + }, + "widgets_values": [ + "{\"model_cls\": \"wan2.1_audio\", \"model_path\": \"/mnt/aigc/users/gaopeng1/ComfyUI/models/lightx2v/Wan2.1-R2V721-Audio-14B-720P\", \"task\": \"i2v\", \"mode\": \"infer\", \"infer_steps\": 4, \"seed\": 2308065231, \"sample_guide_scale\": 1.0, \"sample_shift\": 5, \"enable_cfg\": false, \"prompt\": \"\", \"negative_prompt\": \"\", \"target_height\": 720, \"target_width\": 1280, \"target_video_length\": 81, \"fps\": 16, \"vae_stride\": [4, 8, 8], \"patch_size\": [1, 2, 2], \"feature_caching\": \"NoCaching\", \"teacache_thresh\": 0.26, \"coefficients\": null, \"use_ret_steps\": false, \"dit_quant_scheme\": \"bf16\", \"t5_quant_scheme\": \"bf16\", \"clip_quant_scheme\": \"fp16\", \"quant_op\": \"vllm\", \"precision_mode\": \"fp32\", \"dit_quantized_ckpt\": null, \"t5_quantized_ckpt\": null, \"clip_quantized_ckpt\": null, \"mm_config\": {\"mm_type\": \"Default\"}, \"rotary_chunk\": false, \"rotary_chunk_size\": 100, \"clean_cuda_cache\": false, \"torch_compile\": false, \"attention_type\": \"flash_attn3\", \"self_attn_1_type\": \"flash_attn3\", \"cross_attn_1_type\": \"flash_attn3\", \"cross_attn_2_type\": \"flash_attn3\", \"cpu_offload\": false, \"offload_granularity\": \"phase\", \"offload_ratio\": 1.0, \"t5_cpu_offload\": false, \"t5_offload_granularity\": \"model\", \"lazy_load\": false, \"unload_modules\": false, \"use_tiny_vae\": false, \"tiny_vae\": false, \"tiny_vae_path\": null, \"use_tiling_vae\": false, \"lora_path\": null, \"strength_model\": 1.0, \"do_mm_calib\": false, \"parallel_attn_type\": null, \"parallel_vae\": false, \"max_area\": false, \"use_prompt_enhancer\": false, \"text_len\": 512, \"video_duration\": 5.0, \"adaptive_resize\": false, \"_class_name\": \"WanModel\", \"_diffusers_version\": \"0.30.0\", \"dim\": 5120, \"eps\": 1e-06, \"ffn_dim\": 13824, \"freq_dim\": 256, \"in_dim\": 16, \"model_type\": \"i2v\", \"num_heads\": 40, \"num_layers\": 40, \"out_dim\": 16}" + ] + }, + { + "id": 118, + "type": "easy float", + "pos": [1477.1490478515625, -516.4893798828125], + "size": [301.4599914550781, 95.50999450683594], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "float", + "type": "FLOAT", + "links": [88, 89] + } + ], + "title": "output_fps", + "properties": { + "Node name for S&R": "easy float" + }, + "widgets_values": [24.000000000000004] + } + ], + "links": [ + [76, 105, 0, 109, 0, "INFERENCE_CONFIG"], + [78, 109, 0, 110, 0, "COMBINED_CONFIG"], + [79, 107, 0, 110, 1, "IMAGE"], + [80, 109, 0, 111, 0, "*"], + [84, 110, 1, 108, 1, "AUDIO"], + [85, 116, 0, 110, 2, "AUDIO"], + [86, 110, 0, 117, 0, "IMAGE"], + [87, 117, 0, 108, 0, "IMAGE"], + [88, 118, 0, 117, 1, "FLOAT"], + [89, 118, 0, 108, 4, "FLOAT"] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.6209213230591553, + "offset": [-32.91825689016712, 1273.01043993064] + }, + "frontendVersion": "1.23.4", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} diff --git a/examples/wan_i2v_with_distill_lora.json b/examples/wan_i2v_with_distill_lora.json index f9b61ca..21b6e9c 100644 --- a/examples/wan_i2v_with_distill_lora.json +++ b/examples/wan_i2v_with_distill_lora.json @@ -7,8 +7,14 @@ { "id": 109, "type": "LightX2VConfigCombiner", - "pos": [1531.4547119140625, -521.6442260742188], - "size": [239.138671875, 126], + "pos": [ + 1531.4547119140625, + -521.6442260742188 + ], + "size": [ + 239.138671875, + 126 + ], "flags": {}, "order": 3, "mode": 0, @@ -53,7 +59,10 @@ { "name": "combined_config", "type": "COMBINED_CONFIG", - "links": [78, 80] + "links": [ + 78, + 80 + ] } ], "properties": { @@ -64,8 +73,14 @@ { "id": 111, "type": "easy showAnything", - "pos": [1839.9300537109375, -494.4144287109375], - "size": [624.5454711914062, 358.7272644042969], + "pos": [ + 1839.9300537109375, + -494.4144287109375 + ], + "size": [ + 624.5454711914062, + 358.7272644042969 + ], "flags": { "collapsed": true }, @@ -93,48 +108,19 @@ "{\"model_cls\": \"wan2.1\", \"model_path\": \"/mnt/aigc/users/lijiaqi2/ComfyUI/models/lightx2v/Wan2.1-I2V-14B-480P\", \"task\": \"i2v\", \"mode\": \"infer\", \"infer_steps\": 4, \"seed\": 1822974886, \"sample_guide_scale\": 1.0, \"sample_shift\": 8, \"enable_cfg\": false, \"prompt\": \"\", \"negative_prompt\": \"\", \"target_height\": 480, \"target_width\": 480, \"target_video_length\": 81, \"fps\": 16, \"vae_stride\": [4, 8, 8], \"patch_size\": [1, 2, 2], \"feature_caching\": \"NoCaching\", \"teacache_thresh\": 0.26, \"coefficients\": null, \"use_ret_steps\": false, \"dit_quant_scheme\": \"bf16\", \"t5_quant_scheme\": \"bf16\", \"clip_quant_scheme\": \"fp16\", \"quant_op\": \"vllm\", \"precision_mode\": \"fp32\", \"dit_quantized_ckpt\": null, \"t5_quantized_ckpt\": null, \"clip_quantized_ckpt\": null, \"mm_config\": {\"mm_type\": \"Default\"}, \"rotary_chunk\": false, \"rotary_chunk_size\": 100, \"clean_cuda_cache\": false, \"torch_compile\": false, \"attention_type\": \"flash_attn3\", \"self_attn_1_type\": \"flash_attn3\", \"cross_attn_1_type\": \"flash_attn3\", \"cross_attn_2_type\": \"flash_attn3\", \"cpu_offload\": false, \"offload_granularity\": \"phase\", \"offload_ratio\": 1.0, \"t5_cpu_offload\": false, \"t5_offload_granularity\": \"model\", \"lazy_load\": false, \"unload_modules\": false, \"use_tiny_vae\": false, \"tiny_vae\": false, \"tiny_vae_path\": null, \"use_tiling_vae\": false, \"lora_path\": null, \"strength_model\": 1.0, \"do_mm_calib\": false, \"parallel_attn_type\": null, \"parallel_vae\": false, \"max_area\": false, \"use_prompt_enhancer\": false, \"text_len\": 512, \"denoising_step_list\": [999, 750, 500, 250], \"_class_name\": \"WanModel\", \"_diffusers_version\": \"0.30.0\", \"dim\": 5120, \"eps\": 1e-06, \"ffn_dim\": 13824, \"freq_dim\": 256, \"in_dim\": 36, \"model_type\": \"i2v\", \"num_heads\": 40, \"num_layers\": 40, \"out_dim\": 16, \"lora_configs\": [{\"path\": \"/mnt/aigc/users/lijiaqi2/ComfyUI/models/lightx2v/loras/Wan21_T2V_14B_lightx2v_cfg_step_distill_lora_rank32.safetensors\", \"strength\": 1.0}]}" ] }, - { - "id": 105, - "type": "LightX2VInferenceConfig", - "pos": [1073.62646484375, -580.9854125976562], - "size": [270, 346], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "inference_config", - "type": "INFERENCE_CONFIG", - "links": [76] - } - ], - "properties": { - "Node name for S&R": "LightX2VInferenceConfig" - }, - "widgets_values": [ - "wan2.1", - "Wan2.1-I2V-14B-480P", - "i2v", - 4, - 1822974886, - "fixed", - 1, - 8, - 480, - 480, - 81, - 16, - "999, 750, 500, 250" - ] - }, { "id": 112, "type": "LightX2VLoRALoader", - "pos": [1083.3714599609375, -178.75082397460938], - "size": [270, 82], + "pos": [ + 1083.3714599609375, + -178.75082397460938 + ], + "size": [ + 270, + 82 + ], "flags": {}, - "order": 1, + "order": 0, "mode": 0, "inputs": [ { @@ -148,7 +134,9 @@ { "name": "lora_chain", "type": "LORA_CHAIN", - "links": [82] + "links": [ + 82 + ] } ], "properties": { @@ -162,17 +150,25 @@ { "id": 107, "type": "LoadImage", - "pos": [1082.250244140625, -34.84281921386719], - "size": [274.080078125, 314.0000305175781], + "pos": [ + 1082.250244140625, + -34.84281921386719 + ], + "size": [ + 274.080078125, + 314.0000305175781 + ], "flags": {}, - "order": 2, + "order": 1, "mode": 0, "inputs": [], "outputs": [ { "name": "IMAGE", "type": "IMAGE", - "links": [79] + "links": [ + 79 + ] }, { "name": "MASK", @@ -183,13 +179,22 @@ "properties": { "Node name for S&R": "LoadImage" }, - "widgets_values": ["00.jpg", "image"] + "widgets_values": [ + "00.jpg", + "image" + ] }, { "id": 110, "type": "LightX2VModularInference", - "pos": [1479.8463134765625, -292.079345703125], - "size": [400, 200], + "pos": [ + 1479.8463134765625, + -292.079345703125 + ], + "size": [ + 400, + 200 + ], "flags": {}, "order": 4, "mode": 0, @@ -216,19 +221,35 @@ { "name": "images", "type": "IMAGE", - "links": [81] + "links": [ + 81 + ] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null } ], "properties": { "Node name for S&R": "LightX2VModularInference" }, - "widgets_values": ["太空漫步,往前跑。 ", ""] + "widgets_values": [ + "太空漫步,往前跑。 ", + "" + ] }, { "id": 108, "type": "VHS_VideoCombine", - "pos": [1940.93017578125, -293.4468688964844], - "size": [220.5830078125, 524.5830078125], + "pos": [ + 1940.93017578125, + -293.4468688964844 + ], + "size": [ + 220.5830078125, + 334 + ], "flags": {}, "order": 6, "mode": 0, @@ -275,6 +296,7 @@ "pix_fmt": "yuv420p", "crf": 19, "save_metadata": true, + "trim_to_audio": false, "pingpong": false, "save_output": true, "videopreview": { @@ -290,24 +312,116 @@ "muted": false } } + }, + { + "id": 105, + "type": "LightX2VInferenceConfig", + "pos": [ + 1073.62646484375, + -580.9854125976562 + ], + "size": [ + 270, + 346 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "inference_config", + "type": "INFERENCE_CONFIG", + "links": [ + 76 + ] + } + ], + "properties": { + "Node name for S&R": "LightX2VInferenceConfig" + }, + "widgets_values": [ + "wan2.1", + "Wan2.1-I2V-14B-480P", + "i2v", + 4, + 1822974886, + "fixed", + 1, + 8, + 480, + 480, + 5, + "", + false + ] } ], "links": [ - [76, 105, 0, 109, 0, "INFERENCE_CONFIG"], - [78, 109, 0, 110, 0, "COMBINED_CONFIG"], - [79, 107, 0, 110, 1, "IMAGE"], - [80, 109, 0, 111, 0, "*"], - [81, 110, 0, 108, 0, "IMAGE"], - [82, 112, 0, 109, 5, "LORA_CHAIN"] + [ + 76, + 105, + 0, + 109, + 0, + "INFERENCE_CONFIG" + ], + [ + 78, + 109, + 0, + 110, + 0, + "COMBINED_CONFIG" + ], + [ + 79, + 107, + 0, + 110, + 1, + "IMAGE" + ], + [ + 80, + 109, + 0, + 111, + 0, + "*" + ], + [ + 81, + 110, + 0, + 108, + 0, + "IMAGE" + ], + [ + 82, + 112, + 0, + 109, + 5, + "LORA_CHAIN" + ] ], "groups": [], "config": {}, "extra": { "ds": { - "scale": 0.7513148009015777, - "offset": [-282.5313156431909, 675.8508927908888] + "scale": 0.9090909090909091, + "offset": [ + -410.4301352345007, + 742.4904101052797 + ] }, - "frontendVersion": "1.19.9" + "frontendVersion": "1.23.4", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true }, "version": 0.4 -} +} \ No newline at end of file diff --git a/examples/wan_t2v_with_distill_lora.json b/examples/wan_t2v_with_distill_lora.json index df2c178..5303094 100644 --- a/examples/wan_t2v_with_distill_lora.json +++ b/examples/wan_t2v_with_distill_lora.json @@ -7,8 +7,14 @@ { "id": 109, "type": "LightX2VConfigCombiner", - "pos": [1531.4547119140625, -521.6442260742188], - "size": [239.138671875, 126], + "pos": [ + 1531.4547119140625, + -521.6442260742188 + ], + "size": [ + 239.138671875, + 126 + ], "flags": {}, "order": 3, "mode": 0, @@ -53,7 +59,10 @@ { "name": "combined_config", "type": "COMBINED_CONFIG", - "links": [78, 80] + "links": [ + 78, + 80 + ] } ], "properties": { @@ -61,72 +70,17 @@ }, "widgets_values": [] }, - { - "id": 107, - "type": "LoadImage", - "pos": [1501.387939453125, 272.5443420410156], - "size": [274.080078125, 314.0000305175781], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [79] - }, - { - "name": "MASK", - "type": "MASK", - "links": null - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": ["00.jpg", "image"] - }, - { - "id": 105, - "type": "LightX2VInferenceConfig", - "pos": [1073.62646484375, -580.9854125976562], - "size": [270, 346], - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "inference_config", - "type": "INFERENCE_CONFIG", - "links": [76] - } - ], - "properties": { - "Node name for S&R": "LightX2VInferenceConfig" - }, - "widgets_values": [ - "wan2.1", - "Wan2.1-T2V-14B", - "t2v", - 4, - 1822974886, - "fixed", - 1, - 8, - 480, - 480, - 81, - 16, - "999, 750, 500, 250" - ] - }, { "id": 111, "type": "easy showAnything", - "pos": [1839.9300537109375, -494.4144287109375], - "size": [624.5454711914062, 358.7272644042969], + "pos": [ + 1839.9300537109375, + -494.4144287109375 + ], + "size": [ + 624.5454711914062, + 358.7272644042969 + ], "flags": { "collapsed": true }, @@ -157,8 +111,14 @@ { "id": 110, "type": "LightX2VModularInference", - "pos": [1496.8944091796875, -278.5135498046875], - "size": [400, 200], + "pos": [ + 1496.8944091796875, + -278.5135498046875 + ], + "size": [ + 400, + 200 + ], "flags": {}, "order": 4, "mode": 0, @@ -185,19 +145,35 @@ { "name": "images", "type": "IMAGE", - "links": [81] + "links": [ + 81 + ] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null } ], "properties": { "Node name for S&R": "LightX2VModularInference" }, - "widgets_values": ["好奇的小兔子。 ", ""] + "widgets_values": [ + "好奇的小兔子。 ", + "" + ] }, { "id": 108, "type": "VHS_VideoCombine", - "pos": [1943.5921630859375, -272.15087890625], - "size": [220.5830078125, 524.5830078125], + "pos": [ + 1943.5921630859375, + -272.15087890625 + ], + "size": [ + 220.5830078125, + 334 + ], "flags": {}, "order": 6, "mode": 0, @@ -244,6 +220,7 @@ "pix_fmt": "yuv420p", "crf": 19, "save_metadata": true, + "trim_to_audio": false, "pingpong": false, "save_output": true, "videopreview": { @@ -263,10 +240,16 @@ { "id": 112, "type": "LightX2VLoRALoader", - "pos": [1068.7301025390625, -174.75782775878906], - "size": [270, 82], + "pos": [ + 1068.7301025390625, + -174.75782775878906 + ], + "size": [ + 270, + 82 + ], "flags": {}, - "order": 2, + "order": 0, "mode": 0, "inputs": [ { @@ -280,7 +263,9 @@ { "name": "lora_chain", "type": "LORA_CHAIN", - "links": [82] + "links": [ + 82 + ] } ], "properties": { @@ -290,24 +275,153 @@ "Wan21_T2V_14B_lightx2v_cfg_step_distill_lora_rank32.safetensors", 1 ] + }, + { + "id": 105, + "type": "LightX2VInferenceConfig", + "pos": [ + 1073.62646484375, + -580.9854125976562 + ], + "size": [ + 270, + 346 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "inference_config", + "type": "INFERENCE_CONFIG", + "links": [ + 76 + ] + } + ], + "properties": { + "Node name for S&R": "LightX2VInferenceConfig" + }, + "widgets_values": [ + "wan2.1", + "Wan2.1-T2V-14B", + "t2v", + 4, + 1822974886, + "fixed", + 1, + 8, + 480, + 480, + 5, + "", + false + ] + }, + { + "id": 107, + "type": "LoadImage", + "pos": [ + 1074.7064208984375, + -30.950040817260742 + ], + "size": [ + 274.080078125, + 314.0000305175781 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 79 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "00.jpg", + "image" + ] } ], "links": [ - [76, 105, 0, 109, 0, "INFERENCE_CONFIG"], - [78, 109, 0, 110, 0, "COMBINED_CONFIG"], - [79, 107, 0, 110, 1, "IMAGE"], - [80, 109, 0, 111, 0, "*"], - [81, 110, 0, 108, 0, "IMAGE"], - [82, 112, 0, 109, 5, "LORA_CHAIN"] + [ + 76, + 105, + 0, + 109, + 0, + "INFERENCE_CONFIG" + ], + [ + 78, + 109, + 0, + 110, + 0, + "COMBINED_CONFIG" + ], + [ + 79, + 107, + 0, + 110, + 1, + "IMAGE" + ], + [ + 80, + 109, + 0, + 111, + 0, + "*" + ], + [ + 81, + 110, + 0, + 108, + 0, + "IMAGE" + ], + [ + 82, + 112, + 0, + 109, + 5, + "LORA_CHAIN" + ] ], "groups": [], "config": {}, "extra": { "ds": { "scale": 0.9090909090909091, - "offset": [-335.01067565958357, 635.6502922171578] + "offset": [ + -509.5614756595836, + 590.1535922171577 + ] }, - "frontendVersion": "1.19.9" + "frontendVersion": "1.23.4", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true }, "version": 0.4 -} +} \ No newline at end of file diff --git a/lightx2v b/lightx2v index 6678847..305350a 160000 --- a/lightx2v +++ b/lightx2v @@ -1 +1 @@ -Subproject commit 66788474897655c9fabc16b43e0ff48f99ed5934 +Subproject commit 305350a16b9facdb79e6cc1ed1a5594312d5f92b diff --git a/nodes.py b/nodes.py index 314161d..9460174 100644 --- a/nodes.py +++ b/nodes.py @@ -98,22 +98,14 @@ class LightX2VInferenceConfig: "tooltip": "Video width", }, ), - "video_length": ( - "INT", + "duration": ( + "FLOAT", { - "default": 81, - "min": 16, - "max": 120, - "tooltip": "Video frame count", - }, - ), - "fps": ( - "INT", - { - "default": 16, - "min": 8, - "max": 30, - "tooltip": "Model output frame rate (cannot be changed)", + "default": 5.0, + "min": 1.0, + "max": 10.0, + "step": 0.1, + "tooltip": "Video duration in seconds", }, ), }, @@ -125,6 +117,13 @@ class LightX2VInferenceConfig: "tooltip": "Custom denoising steps for distillation models (comma-separated, e.g., '999,750,500,250'). Leave empty to use model defaults.", }, ), + "adaptive_resize": ( + "BOOLEAN", + { + "default": False, + "tooltip": "Adaptive resize input image to target aspect ratio", + }, + ), }, } @@ -144,12 +143,30 @@ class LightX2VInferenceConfig: sample_shift, height, width, - video_length, - fps, + duration, denoising_steps="", + adaptive_resize=False, ): """Create basic inference configuration.""" model_path = get_model_full_path(model_name) + + if model_cls == "hunyuan": + fps = 24 + else: + fps = 16 + + video_length = int(round(duration * fps)) + + if video_length < 16: + video_length = 16 + + remainder = (video_length - 1) % 4 + if remainder != 0: + video_length = video_length + (4 - remainder) + + #TODO(xxx): + if "wan2.1_audio" in [model_cls]: + video_length = 81 config = { "model_cls": model_cls, @@ -163,6 +180,8 @@ class LightX2VInferenceConfig: "width": width, "video_length": video_length, "fps": fps, + "video_duration": duration, + "adaptive_resize": adaptive_resize, } if denoising_steps and denoising_steps.strip(): @@ -579,8 +598,8 @@ class LightX2VModularInference: }, } - RETURN_TYPES = ("IMAGE",) - RETURN_NAMES = ("images",) + RETURN_TYPES = ("IMAGE", "AUDIO") + RETURN_NAMES = ("images", "audio") FUNCTION = "generate" CATEGORY = "LightX2V/Inference" @@ -634,21 +653,36 @@ class LightX2VModularInference: pil_image.save(tmp.name) config.image_path = tmp.name temp_files.append(tmp.name) + logging.info(f"Image saved to {tmp.name}") if ( audio is not None and hasattr(config, "model_cls") and "audio" in config.model_cls ): - if isinstance(audio, tuple) and len(audio) == 2: + if isinstance(audio, dict) and 'waveform' in audio and 'sample_rate' in audio: + waveform = audio['waveform'] + sample_rate = audio['sample_rate'] + + # Handle different waveform shapes + if isinstance(waveform, torch.Tensor): + if waveform.dim() == 3: # [batch, channels, samples] + waveform = waveform[0] # Take first batch + if waveform.dim() == 2: # [channels, samples] + # Convert to [samples, channels] for wav file + waveform = waveform.transpose(0, 1) + waveform = waveform.cpu().numpy() + elif isinstance(audio, tuple) and len(audio) == 2: + # Legacy format support waveform, sample_rate = audio - if isinstance(waveform, torch.Tensor): waveform = waveform.cpu().numpy() + else: + raise ValueError(f"Unsupported audio format: {type(audio)}") - with tempfile.NamedTemporaryFile( - suffix=".wav", delete=False - ) as tmp: + with tempfile.NamedTemporaryFile( + suffix=".wav", delete=False + ) as tmp: try: import scipy.io.wavfile as wavfile except ImportError: @@ -674,6 +708,9 @@ class LightX2VModularInference: config.audio_path = tmp.name temp_files.append(tmp.name) + logging.info(f"Audio saved to {tmp.name}") + + config_hash = self._get_config_hash(config) needs_reinit = ( self._current_runner is None @@ -694,7 +731,7 @@ class LightX2VModularInference: self._current_runner.config = config total_steps = getattr(config, "infer_steps", 40) - progress = ProgressBar(total_steps) + progress = ProgressBar(100) def update_progress(current_step, total): progress.update_absolute(current_step) @@ -702,11 +739,9 @@ class LightX2VModularInference: if hasattr(self._current_runner, "set_progress_callback"): self._current_runner.set_progress_callback(update_progress) - if hasattr(self._current_runner, "run_pipeline"): - images = self._current_runner.run_pipeline(save_video=False) - else: - images = self._current_runner() - + + images, audio = self._current_runner.run_pipeline(save_video=False) + if getattr(config, "unload_after_inference", False): del self._current_runner self._current_runner = None @@ -715,11 +750,8 @@ class LightX2VModularInference: torch.cuda.empty_cache() gc.collect() - images = (images + 1) / 2 - images = images.squeeze(0).permute(1, 2, 3, 0).cpu() - images = torch.clamp(images, 0, 1) - return (images,) + return (images, audio) except Exception as e: logging.error(f"Error during inference: {e}")