feat: enhance ModularConfigManager and LightX2VInferenceConfig with new video duration and adaptive resize options; update inference handling to support audio output

This commit is contained in:
gaclove
2025-07-24 18:34:23 +08:00
parent b11310970c
commit 022ddfd619
7 changed files with 899 additions and 235 deletions
+6
View File
@@ -324,6 +324,12 @@ class ModularConfigManager:
if "fps" in config:
updates["fps"] = config["fps"]
if "video_duration" in config:
updates["video_duration"] = config["video_duration"]
if "adaptive_resize" in config:
updates["adaptive_resize"] = config["adaptive_resize"]
if "denoising_step_list" in config:
updates["denoising_step_list"] = config["denoising_step_list"]
+60 -50
View File
@@ -107,49 +107,6 @@
},
"widgets_values": []
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [
1073.62646484375,
-580.9854125976562
],
"size": [
270,
346
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [
76
]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1",
"Wan2.1-I2V-14B-480P",
"i2v",
40,
1822974886,
"fixed",
5,
3,
480,
480,
81,
16,
""
]
},
{
"id": 110,
"type": "LightX2VModularInference",
@@ -190,6 +147,11 @@
"links": [
81
]
},
{
"name": "audio",
"type": "AUDIO",
"links": null
}
],
"properties": {
@@ -209,7 +171,7 @@
],
"size": [
220.5830078125,
524.5830078125
334
],
"flags": {},
"order": 5,
@@ -257,6 +219,7 @@
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"videopreview": {
@@ -277,8 +240,8 @@
"id": 111,
"type": "easy showAnything",
"pos": [
1773.56689453125,
-706.2327270507812
1766.246337890625,
-545.1817016601562
],
"size": [
624.5454711914062,
@@ -310,6 +273,49 @@
"widgets_values": [
"{\"model_cls\": \"wan2.1\", \"model_path\": \"/mnt/aigc/users/lijiaqi2/ComfyUI/models/lightx2v/Wan2.1-I2V-14B-480P\", \"task\": \"i2v\", \"mode\": \"infer\", \"infer_steps\": 40, \"seed\": 1822974886, \"sample_guide_scale\": 5.0, \"sample_shift\": 3, \"enable_cfg\": true, \"prompt\": \"\", \"negative_prompt\": \"\", \"target_height\": 480, \"target_width\": 480, \"target_video_length\": 81, \"fps\": 16, \"vae_stride\": [4, 8, 8], \"patch_size\": [1, 2, 2], \"feature_caching\": \"NoCaching\", \"teacache_thresh\": 0.26, \"coefficients\": null, \"use_ret_steps\": false, \"dit_quant_scheme\": \"bf16\", \"t5_quant_scheme\": \"bf16\", \"clip_quant_scheme\": \"fp16\", \"quant_op\": \"vllm\", \"precision_mode\": \"fp32\", \"dit_quantized_ckpt\": null, \"t5_quantized_ckpt\": null, \"clip_quantized_ckpt\": null, \"mm_config\": {\"mm_type\": \"Default\"}, \"rotary_chunk\": false, \"rotary_chunk_size\": 100, \"clean_cuda_cache\": false, \"torch_compile\": false, \"attention_type\": \"flash_attn3\", \"self_attn_1_type\": \"flash_attn3\", \"cross_attn_1_type\": \"flash_attn3\", \"cross_attn_2_type\": \"flash_attn3\", \"cpu_offload\": false, \"offload_granularity\": \"phase\", \"offload_ratio\": 1.0, \"t5_cpu_offload\": false, \"t5_offload_granularity\": \"model\", \"lazy_load\": false, \"unload_modules\": false, \"use_tiny_vae\": false, \"tiny_vae\": false, \"tiny_vae_path\": null, \"use_tiling_vae\": false, \"lora_path\": null, \"strength_model\": 1.0, \"do_mm_calib\": false, \"parallel_attn_type\": null, \"parallel_vae\": false, \"max_area\": false, \"use_prompt_enhancer\": false, \"text_len\": 512, \"_class_name\": \"WanModel\", \"_diffusers_version\": \"0.30.0\", \"dim\": 5120, \"eps\": 1e-06, \"ffn_dim\": 13824, \"freq_dim\": 256, \"in_dim\": 36, \"model_type\": \"i2v\", \"num_heads\": 40, \"num_layers\": 40, \"out_dim\": 16}"
]
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [
1073.62646484375,
-580.9854125976562
],
"size": [
270,
346
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [
76
]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1",
"Wan2.1-I2V-14B-480P",
"i2v",
40,
1822974886,
"fixed",
5,
3,
480,
480,
5,
"",
""
]
}
],
"links": [
@@ -360,11 +366,15 @@
"ds": {
"scale": 1,
"offset": [
-697.2225740954825,
708.7013290998841
-584.0960104591186,
687.7305563726111
]
},
"frontendVersion": "1.19.9"
"frontendVersion": "1.23.4",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
}
+388
View File
@@ -0,0 +1,388 @@
{
"id": "8e881096-4f7b-4633-b34b-6b9c0f8aa093",
"revision": 0,
"last_node_id": 118,
"last_link_id": 89,
"nodes": [
{
"id": 109,
"type": "LightX2VConfigCombiner",
"pos": [1090.0107421875, -940.6998901367188],
"size": [239.138671875, 126],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"link": 76
},
{
"name": "teacache_config",
"shape": 7,
"type": "TEACACHE_CONFIG",
"link": null
},
{
"name": "quantization_config",
"shape": 7,
"type": "QUANT_CONFIG",
"link": null
},
{
"name": "memory_config",
"shape": 7,
"type": "MEMORY_CONFIG",
"link": null
},
{
"name": "vae_config",
"shape": 7,
"type": "VAE_CONFIG",
"link": null
},
{
"name": "lora_chain",
"shape": 7,
"type": "LORA_CHAIN",
"link": null
}
],
"outputs": [
{
"name": "combined_config",
"type": "COMBINED_CONFIG",
"links": [78, 80]
}
],
"properties": {
"Node name for S&R": "LightX2VConfigCombiner"
},
"widgets_values": []
},
{
"id": 110,
"type": "LightX2VModularInference",
"pos": [1451.4600830078125, -804.5155029296875],
"size": [400, 200],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "combined_config",
"type": "COMBINED_CONFIG",
"link": 78
},
{
"name": "image",
"shape": 7,
"type": "IMAGE",
"link": 79
},
{
"name": "audio",
"shape": 7,
"type": "AUDIO",
"link": 85
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [86]
},
{
"name": "audio",
"type": "AUDIO",
"links": [84]
}
],
"properties": {
"Node name for S&R": "LightX2VModularInference"
},
"widgets_values": [
"The video features a old lady is saying something and knitting a sweater.",
"色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走"
]
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [775.7962646484375, -936.5638427734375],
"size": [270, 346],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [76]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1_audio",
"Wan2.1-R2V721-Audio-14B-720P",
"i2v",
4,
407287721,
"randomize",
1,
5,
720,
1280,
5,
"",
""
]
},
{
"id": 117,
"type": "RIFEInterpolation",
"pos": [1999.54443359375, -796.881591796875],
"size": [270, 130],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 86
},
{
"name": "target_fps",
"type": "FLOAT",
"widget": {
"name": "target_fps"
},
"link": 88
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [87]
}
],
"properties": {
"Node name for S&R": "RIFEInterpolation"
},
"widgets_values": [16, 24, 1, "flownet.pkl"]
},
{
"id": 108,
"type": "VHS_VideoCombine",
"pos": [2351.4580078125, -794.34375],
"size": [220.5830078125, 496.25701904296875],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 87
},
{
"name": "audio",
"shape": 7,
"type": "AUDIO",
"link": 84
},
{
"name": "meta_batch",
"shape": 7,
"type": "VHS_BatchManager",
"link": null
},
{
"name": "vae",
"shape": 7,
"type": "VAE",
"link": null
},
{
"name": "frame_rate",
"type": "FLOAT",
"widget": {
"name": "frame_rate"
},
"link": 89
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "AnimateDiff",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "AnimateDiff_00013-audio.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24,
"workflow": "AnimateDiff_00013.png",
"fullpath": "/mnt/aigc/users/gaopeng1/ComfyUI/output/AnimateDiff_00013-audio.mp4"
},
"muted": false
}
}
},
{
"id": 107,
"type": "LoadImage",
"pos": [1041.3621826171875, -540.7197265625],
"size": [274.080078125, 314.0000305175781],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [79]
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": ["15.png", "image"]
},
{
"id": 116,
"type": "LoadAudio",
"pos": [1034.8560791015625, -166.87115478515625],
"size": [270, 136],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [85]
}
],
"properties": {
"Node name for S&R": "LoadAudio"
},
"widgets_values": ["15.wav", null, null]
},
{
"id": 111,
"type": "easy showAnything",
"pos": [1471.281982421875, -908.9144897460938],
"size": [624.5454711914062, 358.7272644042969],
"flags": {
"collapsed": true
},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "anything",
"shape": 7,
"type": "*",
"link": 80
}
],
"outputs": [
{
"name": "output",
"type": "*",
"links": null
}
],
"title": "config",
"properties": {
"Node name for S&R": "easy showAnything"
},
"widgets_values": [
"{\"model_cls\": \"wan2.1_audio\", \"model_path\": \"/mnt/aigc/users/gaopeng1/ComfyUI/models/lightx2v/Wan2.1-R2V721-Audio-14B-720P\", \"task\": \"i2v\", \"mode\": \"infer\", \"infer_steps\": 4, \"seed\": 2308065231, \"sample_guide_scale\": 1.0, \"sample_shift\": 5, \"enable_cfg\": false, \"prompt\": \"\", \"negative_prompt\": \"\", \"target_height\": 720, \"target_width\": 1280, \"target_video_length\": 81, \"fps\": 16, \"vae_stride\": [4, 8, 8], \"patch_size\": [1, 2, 2], \"feature_caching\": \"NoCaching\", \"teacache_thresh\": 0.26, \"coefficients\": null, \"use_ret_steps\": false, \"dit_quant_scheme\": \"bf16\", \"t5_quant_scheme\": \"bf16\", \"clip_quant_scheme\": \"fp16\", \"quant_op\": \"vllm\", \"precision_mode\": \"fp32\", \"dit_quantized_ckpt\": null, \"t5_quantized_ckpt\": null, \"clip_quantized_ckpt\": null, \"mm_config\": {\"mm_type\": \"Default\"}, \"rotary_chunk\": false, \"rotary_chunk_size\": 100, \"clean_cuda_cache\": false, \"torch_compile\": false, \"attention_type\": \"flash_attn3\", \"self_attn_1_type\": \"flash_attn3\", \"cross_attn_1_type\": \"flash_attn3\", \"cross_attn_2_type\": \"flash_attn3\", \"cpu_offload\": false, \"offload_granularity\": \"phase\", \"offload_ratio\": 1.0, \"t5_cpu_offload\": false, \"t5_offload_granularity\": \"model\", \"lazy_load\": false, \"unload_modules\": false, \"use_tiny_vae\": false, \"tiny_vae\": false, \"tiny_vae_path\": null, \"use_tiling_vae\": false, \"lora_path\": null, \"strength_model\": 1.0, \"do_mm_calib\": false, \"parallel_attn_type\": null, \"parallel_vae\": false, \"max_area\": false, \"use_prompt_enhancer\": false, \"text_len\": 512, \"video_duration\": 5.0, \"adaptive_resize\": false, \"_class_name\": \"WanModel\", \"_diffusers_version\": \"0.30.0\", \"dim\": 5120, \"eps\": 1e-06, \"ffn_dim\": 13824, \"freq_dim\": 256, \"in_dim\": 16, \"model_type\": \"i2v\", \"num_heads\": 40, \"num_layers\": 40, \"out_dim\": 16}"
]
},
{
"id": 118,
"type": "easy float",
"pos": [1477.1490478515625, -516.4893798828125],
"size": [301.4599914550781, 95.50999450683594],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "float",
"type": "FLOAT",
"links": [88, 89]
}
],
"title": "output_fps",
"properties": {
"Node name for S&R": "easy float"
},
"widgets_values": [24.000000000000004]
}
],
"links": [
[76, 105, 0, 109, 0, "INFERENCE_CONFIG"],
[78, 109, 0, 110, 0, "COMBINED_CONFIG"],
[79, 107, 0, 110, 1, "IMAGE"],
[80, 109, 0, 111, 0, "*"],
[84, 110, 1, 108, 1, "AUDIO"],
[85, 116, 0, 110, 2, "AUDIO"],
[86, 110, 0, 117, 0, "IMAGE"],
[87, 117, 0, 108, 0, "IMAGE"],
[88, 118, 0, 117, 1, "FLOAT"],
[89, 118, 0, 108, 4, "FLOAT"]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.6209213230591553,
"offset": [-32.91825689016712, 1273.01043993064]
},
"frontendVersion": "1.23.4",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
+179 -65
View File
@@ -7,8 +7,14 @@
{
"id": 109,
"type": "LightX2VConfigCombiner",
"pos": [1531.4547119140625, -521.6442260742188],
"size": [239.138671875, 126],
"pos": [
1531.4547119140625,
-521.6442260742188
],
"size": [
239.138671875,
126
],
"flags": {},
"order": 3,
"mode": 0,
@@ -53,7 +59,10 @@
{
"name": "combined_config",
"type": "COMBINED_CONFIG",
"links": [78, 80]
"links": [
78,
80
]
}
],
"properties": {
@@ -64,8 +73,14 @@
{
"id": 111,
"type": "easy showAnything",
"pos": [1839.9300537109375, -494.4144287109375],
"size": [624.5454711914062, 358.7272644042969],
"pos": [
1839.9300537109375,
-494.4144287109375
],
"size": [
624.5454711914062,
358.7272644042969
],
"flags": {
"collapsed": true
},
@@ -93,48 +108,19 @@
"{\"model_cls\": \"wan2.1\", \"model_path\": \"/mnt/aigc/users/lijiaqi2/ComfyUI/models/lightx2v/Wan2.1-I2V-14B-480P\", \"task\": \"i2v\", \"mode\": \"infer\", \"infer_steps\": 4, \"seed\": 1822974886, \"sample_guide_scale\": 1.0, \"sample_shift\": 8, \"enable_cfg\": false, \"prompt\": \"\", \"negative_prompt\": \"\", \"target_height\": 480, \"target_width\": 480, \"target_video_length\": 81, \"fps\": 16, \"vae_stride\": [4, 8, 8], \"patch_size\": [1, 2, 2], \"feature_caching\": \"NoCaching\", \"teacache_thresh\": 0.26, \"coefficients\": null, \"use_ret_steps\": false, \"dit_quant_scheme\": \"bf16\", \"t5_quant_scheme\": \"bf16\", \"clip_quant_scheme\": \"fp16\", \"quant_op\": \"vllm\", \"precision_mode\": \"fp32\", \"dit_quantized_ckpt\": null, \"t5_quantized_ckpt\": null, \"clip_quantized_ckpt\": null, \"mm_config\": {\"mm_type\": \"Default\"}, \"rotary_chunk\": false, \"rotary_chunk_size\": 100, \"clean_cuda_cache\": false, \"torch_compile\": false, \"attention_type\": \"flash_attn3\", \"self_attn_1_type\": \"flash_attn3\", \"cross_attn_1_type\": \"flash_attn3\", \"cross_attn_2_type\": \"flash_attn3\", \"cpu_offload\": false, \"offload_granularity\": \"phase\", \"offload_ratio\": 1.0, \"t5_cpu_offload\": false, \"t5_offload_granularity\": \"model\", \"lazy_load\": false, \"unload_modules\": false, \"use_tiny_vae\": false, \"tiny_vae\": false, \"tiny_vae_path\": null, \"use_tiling_vae\": false, \"lora_path\": null, \"strength_model\": 1.0, \"do_mm_calib\": false, \"parallel_attn_type\": null, \"parallel_vae\": false, \"max_area\": false, \"use_prompt_enhancer\": false, \"text_len\": 512, \"denoising_step_list\": [999, 750, 500, 250], \"_class_name\": \"WanModel\", \"_diffusers_version\": \"0.30.0\", \"dim\": 5120, \"eps\": 1e-06, \"ffn_dim\": 13824, \"freq_dim\": 256, \"in_dim\": 36, \"model_type\": \"i2v\", \"num_heads\": 40, \"num_layers\": 40, \"out_dim\": 16, \"lora_configs\": [{\"path\": \"/mnt/aigc/users/lijiaqi2/ComfyUI/models/lightx2v/loras/Wan21_T2V_14B_lightx2v_cfg_step_distill_lora_rank32.safetensors\", \"strength\": 1.0}]}"
]
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [1073.62646484375, -580.9854125976562],
"size": [270, 346],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [76]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1",
"Wan2.1-I2V-14B-480P",
"i2v",
4,
1822974886,
"fixed",
1,
8,
480,
480,
81,
16,
"999, 750, 500, 250"
]
},
{
"id": 112,
"type": "LightX2VLoRALoader",
"pos": [1083.3714599609375, -178.75082397460938],
"size": [270, 82],
"pos": [
1083.3714599609375,
-178.75082397460938
],
"size": [
270,
82
],
"flags": {},
"order": 1,
"order": 0,
"mode": 0,
"inputs": [
{
@@ -148,7 +134,9 @@
{
"name": "lora_chain",
"type": "LORA_CHAIN",
"links": [82]
"links": [
82
]
}
],
"properties": {
@@ -162,17 +150,25 @@
{
"id": 107,
"type": "LoadImage",
"pos": [1082.250244140625, -34.84281921386719],
"size": [274.080078125, 314.0000305175781],
"pos": [
1082.250244140625,
-34.84281921386719
],
"size": [
274.080078125,
314.0000305175781
],
"flags": {},
"order": 2,
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [79]
"links": [
79
]
},
{
"name": "MASK",
@@ -183,13 +179,22 @@
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": ["00.jpg", "image"]
"widgets_values": [
"00.jpg",
"image"
]
},
{
"id": 110,
"type": "LightX2VModularInference",
"pos": [1479.8463134765625, -292.079345703125],
"size": [400, 200],
"pos": [
1479.8463134765625,
-292.079345703125
],
"size": [
400,
200
],
"flags": {},
"order": 4,
"mode": 0,
@@ -216,19 +221,35 @@
{
"name": "images",
"type": "IMAGE",
"links": [81]
"links": [
81
]
},
{
"name": "audio",
"type": "AUDIO",
"links": null
}
],
"properties": {
"Node name for S&R": "LightX2VModularInference"
},
"widgets_values": ["太空漫步,往前跑。 ", ""]
"widgets_values": [
"太空漫步,往前跑。 ",
""
]
},
{
"id": 108,
"type": "VHS_VideoCombine",
"pos": [1940.93017578125, -293.4468688964844],
"size": [220.5830078125, 524.5830078125],
"pos": [
1940.93017578125,
-293.4468688964844
],
"size": [
220.5830078125,
334
],
"flags": {},
"order": 6,
"mode": 0,
@@ -275,6 +296,7 @@
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"videopreview": {
@@ -290,24 +312,116 @@
"muted": false
}
}
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [
1073.62646484375,
-580.9854125976562
],
"size": [
270,
346
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [
76
]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1",
"Wan2.1-I2V-14B-480P",
"i2v",
4,
1822974886,
"fixed",
1,
8,
480,
480,
5,
"",
false
]
}
],
"links": [
[76, 105, 0, 109, 0, "INFERENCE_CONFIG"],
[78, 109, 0, 110, 0, "COMBINED_CONFIG"],
[79, 107, 0, 110, 1, "IMAGE"],
[80, 109, 0, 111, 0, "*"],
[81, 110, 0, 108, 0, "IMAGE"],
[82, 112, 0, 109, 5, "LORA_CHAIN"]
[
76,
105,
0,
109,
0,
"INFERENCE_CONFIG"
],
[
78,
109,
0,
110,
0,
"COMBINED_CONFIG"
],
[
79,
107,
0,
110,
1,
"IMAGE"
],
[
80,
109,
0,
111,
0,
"*"
],
[
81,
110,
0,
108,
0,
"IMAGE"
],
[
82,
112,
0,
109,
5,
"LORA_CHAIN"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015777,
"offset": [-282.5313156431909, 675.8508927908888]
"scale": 0.9090909090909091,
"offset": [
-410.4301352345007,
742.4904101052797
]
},
"frontendVersion": "1.19.9"
"frontendVersion": "1.23.4",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
}
+199 -85
View File
@@ -7,8 +7,14 @@
{
"id": 109,
"type": "LightX2VConfigCombiner",
"pos": [1531.4547119140625, -521.6442260742188],
"size": [239.138671875, 126],
"pos": [
1531.4547119140625,
-521.6442260742188
],
"size": [
239.138671875,
126
],
"flags": {},
"order": 3,
"mode": 0,
@@ -53,7 +59,10 @@
{
"name": "combined_config",
"type": "COMBINED_CONFIG",
"links": [78, 80]
"links": [
78,
80
]
}
],
"properties": {
@@ -61,72 +70,17 @@
},
"widgets_values": []
},
{
"id": 107,
"type": "LoadImage",
"pos": [1501.387939453125, 272.5443420410156],
"size": [274.080078125, 314.0000305175781],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [79]
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": ["00.jpg", "image"]
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [1073.62646484375, -580.9854125976562],
"size": [270, 346],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [76]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1",
"Wan2.1-T2V-14B",
"t2v",
4,
1822974886,
"fixed",
1,
8,
480,
480,
81,
16,
"999, 750, 500, 250"
]
},
{
"id": 111,
"type": "easy showAnything",
"pos": [1839.9300537109375, -494.4144287109375],
"size": [624.5454711914062, 358.7272644042969],
"pos": [
1839.9300537109375,
-494.4144287109375
],
"size": [
624.5454711914062,
358.7272644042969
],
"flags": {
"collapsed": true
},
@@ -157,8 +111,14 @@
{
"id": 110,
"type": "LightX2VModularInference",
"pos": [1496.8944091796875, -278.5135498046875],
"size": [400, 200],
"pos": [
1496.8944091796875,
-278.5135498046875
],
"size": [
400,
200
],
"flags": {},
"order": 4,
"mode": 0,
@@ -185,19 +145,35 @@
{
"name": "images",
"type": "IMAGE",
"links": [81]
"links": [
81
]
},
{
"name": "audio",
"type": "AUDIO",
"links": null
}
],
"properties": {
"Node name for S&R": "LightX2VModularInference"
},
"widgets_values": ["好奇的小兔子。 ", ""]
"widgets_values": [
"好奇的小兔子。 ",
""
]
},
{
"id": 108,
"type": "VHS_VideoCombine",
"pos": [1943.5921630859375, -272.15087890625],
"size": [220.5830078125, 524.5830078125],
"pos": [
1943.5921630859375,
-272.15087890625
],
"size": [
220.5830078125,
334
],
"flags": {},
"order": 6,
"mode": 0,
@@ -244,6 +220,7 @@
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"videopreview": {
@@ -263,10 +240,16 @@
{
"id": 112,
"type": "LightX2VLoRALoader",
"pos": [1068.7301025390625, -174.75782775878906],
"size": [270, 82],
"pos": [
1068.7301025390625,
-174.75782775878906
],
"size": [
270,
82
],
"flags": {},
"order": 2,
"order": 0,
"mode": 0,
"inputs": [
{
@@ -280,7 +263,9 @@
{
"name": "lora_chain",
"type": "LORA_CHAIN",
"links": [82]
"links": [
82
]
}
],
"properties": {
@@ -290,24 +275,153 @@
"Wan21_T2V_14B_lightx2v_cfg_step_distill_lora_rank32.safetensors",
1
]
},
{
"id": 105,
"type": "LightX2VInferenceConfig",
"pos": [
1073.62646484375,
-580.9854125976562
],
"size": [
270,
346
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "inference_config",
"type": "INFERENCE_CONFIG",
"links": [
76
]
}
],
"properties": {
"Node name for S&R": "LightX2VInferenceConfig"
},
"widgets_values": [
"wan2.1",
"Wan2.1-T2V-14B",
"t2v",
4,
1822974886,
"fixed",
1,
8,
480,
480,
5,
"",
false
]
},
{
"id": 107,
"type": "LoadImage",
"pos": [
1074.7064208984375,
-30.950040817260742
],
"size": [
274.080078125,
314.0000305175781
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
79
]
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"00.jpg",
"image"
]
}
],
"links": [
[76, 105, 0, 109, 0, "INFERENCE_CONFIG"],
[78, 109, 0, 110, 0, "COMBINED_CONFIG"],
[79, 107, 0, 110, 1, "IMAGE"],
[80, 109, 0, 111, 0, "*"],
[81, 110, 0, 108, 0, "IMAGE"],
[82, 112, 0, 109, 5, "LORA_CHAIN"]
[
76,
105,
0,
109,
0,
"INFERENCE_CONFIG"
],
[
78,
109,
0,
110,
0,
"COMBINED_CONFIG"
],
[
79,
107,
0,
110,
1,
"IMAGE"
],
[
80,
109,
0,
111,
0,
"*"
],
[
81,
110,
0,
108,
0,
"IMAGE"
],
[
82,
112,
0,
109,
5,
"LORA_CHAIN"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.9090909090909091,
"offset": [-335.01067565958357, 635.6502922171578]
"offset": [
-509.5614756595836,
590.1535922171577
]
},
"frontendVersion": "1.19.9"
"frontendVersion": "1.23.4",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
}
+66 -34
View File
@@ -98,22 +98,14 @@ class LightX2VInferenceConfig:
"tooltip": "Video width",
},
),
"video_length": (
"INT",
"duration": (
"FLOAT",
{
"default": 81,
"min": 16,
"max": 120,
"tooltip": "Video frame count",
},
),
"fps": (
"INT",
{
"default": 16,
"min": 8,
"max": 30,
"tooltip": "Model output frame rate (cannot be changed)",
"default": 5.0,
"min": 1.0,
"max": 10.0,
"step": 0.1,
"tooltip": "Video duration in seconds",
},
),
},
@@ -125,6 +117,13 @@ class LightX2VInferenceConfig:
"tooltip": "Custom denoising steps for distillation models (comma-separated, e.g., '999,750,500,250'). Leave empty to use model defaults.",
},
),
"adaptive_resize": (
"BOOLEAN",
{
"default": False,
"tooltip": "Adaptive resize input image to target aspect ratio",
},
),
},
}
@@ -144,12 +143,30 @@ class LightX2VInferenceConfig:
sample_shift,
height,
width,
video_length,
fps,
duration,
denoising_steps="",
adaptive_resize=False,
):
"""Create basic inference configuration."""
model_path = get_model_full_path(model_name)
if model_cls == "hunyuan":
fps = 24
else:
fps = 16
video_length = int(round(duration * fps))
if video_length < 16:
video_length = 16
remainder = (video_length - 1) % 4
if remainder != 0:
video_length = video_length + (4 - remainder)
#TODO(xxx):
if "wan2.1_audio" in [model_cls]:
video_length = 81
config = {
"model_cls": model_cls,
@@ -163,6 +180,8 @@ class LightX2VInferenceConfig:
"width": width,
"video_length": video_length,
"fps": fps,
"video_duration": duration,
"adaptive_resize": adaptive_resize,
}
if denoising_steps and denoising_steps.strip():
@@ -579,8 +598,8 @@ class LightX2VModularInference:
},
}
RETURN_TYPES = ("IMAGE",)
RETURN_NAMES = ("images",)
RETURN_TYPES = ("IMAGE", "AUDIO")
RETURN_NAMES = ("images", "audio")
FUNCTION = "generate"
CATEGORY = "LightX2V/Inference"
@@ -634,21 +653,36 @@ class LightX2VModularInference:
pil_image.save(tmp.name)
config.image_path = tmp.name
temp_files.append(tmp.name)
logging.info(f"Image saved to {tmp.name}")
if (
audio is not None
and hasattr(config, "model_cls")
and "audio" in config.model_cls
):
if isinstance(audio, tuple) and len(audio) == 2:
if isinstance(audio, dict) and 'waveform' in audio and 'sample_rate' in audio:
waveform = audio['waveform']
sample_rate = audio['sample_rate']
# Handle different waveform shapes
if isinstance(waveform, torch.Tensor):
if waveform.dim() == 3: # [batch, channels, samples]
waveform = waveform[0] # Take first batch
if waveform.dim() == 2: # [channels, samples]
# Convert to [samples, channels] for wav file
waveform = waveform.transpose(0, 1)
waveform = waveform.cpu().numpy()
elif isinstance(audio, tuple) and len(audio) == 2:
# Legacy format support
waveform, sample_rate = audio
if isinstance(waveform, torch.Tensor):
waveform = waveform.cpu().numpy()
else:
raise ValueError(f"Unsupported audio format: {type(audio)}")
with tempfile.NamedTemporaryFile(
suffix=".wav", delete=False
) as tmp:
with tempfile.NamedTemporaryFile(
suffix=".wav", delete=False
) as tmp:
try:
import scipy.io.wavfile as wavfile
except ImportError:
@@ -674,6 +708,9 @@ class LightX2VModularInference:
config.audio_path = tmp.name
temp_files.append(tmp.name)
logging.info(f"Audio saved to {tmp.name}")
config_hash = self._get_config_hash(config)
needs_reinit = (
self._current_runner is None
@@ -694,7 +731,7 @@ class LightX2VModularInference:
self._current_runner.config = config
total_steps = getattr(config, "infer_steps", 40)
progress = ProgressBar(total_steps)
progress = ProgressBar(100)
def update_progress(current_step, total):
progress.update_absolute(current_step)
@@ -702,11 +739,9 @@ class LightX2VModularInference:
if hasattr(self._current_runner, "set_progress_callback"):
self._current_runner.set_progress_callback(update_progress)
if hasattr(self._current_runner, "run_pipeline"):
images = self._current_runner.run_pipeline(save_video=False)
else:
images = self._current_runner()
images, audio = self._current_runner.run_pipeline(save_video=False)
if getattr(config, "unload_after_inference", False):
del self._current_runner
self._current_runner = None
@@ -715,11 +750,8 @@ class LightX2VModularInference:
torch.cuda.empty_cache()
gc.collect()
images = (images + 1) / 2
images = images.squeeze(0).permute(1, 2, 3, 0).cpu()
images = torch.clamp(images, 0, 1)
return (images,)
return (images, audio)
except Exception as e:
logging.error(f"Error during inference: {e}")