diff --git a/InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_24_000_crop_0125.mp4 b/InterDemo/TIAP2V/dwpose/pose1 (2).mp4 similarity index 100% rename from InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_24_000_crop_0125.mp4 rename to InterDemo/TIAP2V/dwpose/pose1 (2).mp4 diff --git a/InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_25_000_crop_0050.mp4 b/InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_25_000_crop_00502.mp4 similarity index 100% rename from InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_25_000_crop_0050.mp4 rename to InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_25_000_crop_00502.mp4 diff --git a/InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_29_000_crop_0025.mp4 b/InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_29_000_crop_00253.mp4 similarity index 100% rename from InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_29_000_crop_0025.mp4 rename to InterDemo/TIAP2V/dwpose/self_collect_20250508_trans_slpit_self_collect_20250508_29_000_crop_00253.mp4 diff --git a/InteractAvatar_node.py b/InteractAvatar_node.py index f6eb975..2c29d57 100644 --- a/InteractAvatar_node.py +++ b/InteractAvatar_node.py @@ -71,7 +71,7 @@ class InteractAvatar_SM_Predata(io.ComfyNode): io.Vae.Input("vae"), io.Image.Input("images"), # image or video io.Image.Input("pose_images"), # image or video - io.Int.Input("short_side", default=512, min=256, max=nodes.MAX_RESOLUTION,step=32,display_mode=io.NumberDisplay.number), + io.Combo.Input("short_side",options= [704,512]), io.String.Input("prompt",multiline=True, default="两只手打招呼 伸出大拇指点赞"), io.String.Input("negative_prompt",multiline=True, default="bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"), io.String.Input("structured_prompt",multiline=True, default=" (Raise both hands slowly) (Wave both hands side to side for greeting),\n (Make a fist) (Extend the thumb upwards)"), diff --git a/README.md b/README.md index d3309c4..33ffe3e 100644 --- a/README.md +++ b/README.md @@ -20,7 +20,10 @@ # ComfyUI_InteractAvatar InteractAvatar is a novel dual-stream DiT framework that enables talking avatars to perform Grounded Human-Object Interaction (GHOI) -# Tips +# Update +* fix bug ,now output video short side muse be 512 or 704 + + * If your Vram <24G,turn on 'offload', ActionAndSong mode use 'long model' and need chocie '2' mode;example img\video\ audio in "InterDemo" dir * test env 64G RAM, 12G VRAM,win11 * The prompt words for the singing mode and the action prompt words must have the same number of lines; @@ -56,6 +59,8 @@ pip install -r requirements.txt ![](https://github.com/smthemex/ComfyUI_InteractAvatar/blob/main/example_workflows/example-song.png) * object ![](https://github.com/smthemex/ComfyUI_InteractAvatar/blob/main/example_workflows/example.png) +* ap2v audio and pose driver +![](https://github.com/smthemex/ComfyUI_InteractAvatar/blob/main/example_workflows/example_ap2v.png) # 5 Citation ``` diff --git a/example_workflows/InteractAvatar.json b/example_workflows/InteractAvatar.json index 33e7137..abd07b8 100644 --- a/example_workflows/InteractAvatar.json +++ b/example_workflows/InteractAvatar.json @@ -1,8 +1,8 @@ { "id": "70258d66-2650-4bb1-b607-75e03c1f7f28", "revision": 0, - "last_node_id": 33, - "last_link_id": 51, + "last_node_id": 70, + "last_link_id": 93, "nodes": [ { "id": 4, @@ -16,8 +16,8 @@ 46 ], "flags": {}, - "order": 21, - "mode": 2, + "order": 33, + "mode": 0, "inputs": [ { "name": "samples", @@ -44,501 +44,6 @@ }, "widgets_values": [] }, - { - "id": 6, - "type": "CreateVideo", - "pos": [ - 21653.483940544334, - -624.0641184307909 - ], - "size": [ - 270, - 78 - ], - "flags": {}, - "order": 23, - "mode": 2, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 5 - }, - { - "name": "audio", - "shape": 7, - "type": "AUDIO", - "link": null - } - ], - "outputs": [ - { - "name": "VIDEO", - "type": "VIDEO", - "links": [ - 6 - ] - } - ], - "properties": { - "Node name for S&R": "CreateVideo" - }, - "widgets_values": [ - 30 - ] - }, - { - "id": 7, - "type": "SaveVideo", - "pos": [ - 21524.550939818026, - -375.9493523272225 - ], - "size": [ - 270, - 106 - ], - "flags": {}, - "order": 25, - "mode": 2, - "inputs": [ - { - "name": "video", - "type": "VIDEO", - "link": 6 - } - ], - "outputs": [], - "properties": {}, - "widgets_values": [ - "video/ComfyUI", - "auto", - "auto" - ] - }, - { - "id": 1, - "type": "InteractAvatar_SM_Model", - "pos": [ - 20655.956740953232, - -826.2880845675337 - ], - "size": [ - 278.662109375, - 178 - ], - "flags": {}, - "order": 0, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "model", - "type": "MODEL", - "links": [ - 13 - ] - } - ], - "properties": { - "Node name for S&R": "InteractAvatar_SM_Model" - }, - "widgets_values": [ - "interact-avatar.safetensors", - "interact-avatar.safetensors", - "none", - "none", - 1 - ] - }, - { - "id": 13, - "type": "InteractAvatar_SM_Sampler", - "pos": [ - 21111.045470959252, - -616.7680274485149 - ], - "size": [ - 281.2984375, - 318 - ], - "flags": {}, - "order": 19, - "mode": 2, - "inputs": [ - { - "name": "model", - "type": "MODEL", - "link": 13 - }, - { - "name": "data_dict", - "type": "CONDITIONING", - "link": 18 - } - ], - "outputs": [ - { - "name": "Latent", - "type": "LATENT", - "links": [ - 15 - ] - } - ], - "properties": { - "Node name for S&R": "InteractAvatar_SM_Sampler" - }, - "widgets_values": [ - 20, - 551863485, - "randomize", - 5, - 5, - 5, - 0, - 25, - 1000, - false - ] - }, - { - "id": 14, - "type": "Note", - "pos": [ - 21073.49649099832, - -891.9647068066879 - ], - "size": [ - 390.28170729224803, - 246.35694222626103 - ], - "flags": {}, - "order": 1, - "mode": 2, - "inputs": [], - "outputs": [], - "properties": {}, - "widgets_values": [ - "mode=a2mv\ntransformer_path=interact-avatar\ntest_data_path=InterDemo/TIA2MV/demo_tia2mv.json\nback_append_frame=1\nframe_num=101\n\n# for pose dirven with audio\nmode=ap2v\ntransformer_path=interact-avatar\ntest_data_path=InterDemo/TIAP2V/demo_ap2v.json\nback_append_frame=1\nframe_num=133\n\nfor song with action\n# mode=a2mv\n# transformer_path=interact-avatar-long\n# test_data_path=demo_song_action.json\n# back_append_frame=2\n# frame_num=1000" - ], - "color": "#432", - "bgcolor": "#653" - }, - { - "id": 8, - "type": "CLIPLoader", - "pos": [ - 20214.19076831883, - -812.6313406695036 - ], - "size": [ - 270, - 106 - ], - "flags": {}, - "order": 2, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "CLIP", - "type": "CLIP", - "links": [ - 16 - ] - } - ], - "properties": { - "Node name for S&R": "CLIPLoader" - }, - "widgets_values": [ - "umt5_xxl_fp8_e4m3fn_scaled.safetensors", - "wan", - "default" - ] - }, - { - "id": 5, - "type": "VAELoader", - "pos": [ - 20224.42956543533, - -644.7584067826607 - ], - "size": [ - 270, - 58 - ], - "flags": {}, - "order": 3, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "VAE", - "type": "VAE", - "links": [ - 4, - 17 - ] - } - ], - "properties": { - "Node name for S&R": "VAELoader" - }, - "widgets_values": [ - "Wan2.2_VAE.pth" - ] - }, - { - "id": 9, - "type": "LoadImage", - "pos": [ - 19817.499679613604, - -756.4208862352102 - ], - "size": [ - 270, - 314 - ], - "flags": {}, - "order": 4, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 19 - ] - }, - { - "name": "MASK", - "type": "MASK", - "links": null - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": [ - "001.jpeg", - "image" - ] - }, - { - "id": 10, - "type": "LoadImage", - "pos": [ - 19839.92752091642, - -392.48311359683214 - ], - "size": [ - 270, - 314 - ], - "flags": {}, - "order": 5, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 20 - ] - }, - { - "name": "MASK", - "type": "MASK", - "links": null - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": [ - "001_obj_handbag.png", - "image" - ] - }, - { - "id": 15, - "type": "InteractAvatar_SM_Predata", - "pos": [ - 20589.88000915876, - -520.7428687917542 - ], - "size": [ - 439.0049413961933, - 585.8462358678404 - ], - "flags": {}, - "order": 16, - "mode": 2, - "inputs": [ - { - "name": "clip", - "type": "CLIP", - "link": 16 - }, - { - "name": "vae", - "type": "VAE", - "link": 17 - }, - { - "name": "images", - "type": "IMAGE", - "link": 19 - }, - { - "name": "pose_images", - "type": "IMAGE", - "link": 20 - }, - { - "name": "audio", - "shape": 7, - "type": "AUDIO", - "link": 24 - }, - { - "name": "object_images", - "shape": 7, - "type": "IMAGE", - "link": 22 - }, - { - "name": "object_mask", - "shape": 7, - "type": "MASK", - "link": 23 - } - ], - "outputs": [ - { - "name": "data_dict", - "type": "CONDITIONING", - "links": [ - 18 - ] - } - ], - "properties": { - "Node name for S&R": "InteractAvatar_SM_Predata" - }, - "widgets_values": [ - 768, - "512", - "The man picks up his handbag.", - "bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards", - "(Left hand holds the handle of the handbag)(Right hand hangs naturally)", - 101, - "a2mv", - 1, - "", - "" - ] - }, - { - "id": 12, - "type": "LoadAudio", - "pos": [ - 20213.269818313645, - -369.5568758206247 - ], - "size": [ - 270, - 136 - ], - "flags": {}, - "order": 6, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 24 - ] - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, - "widgets_values": [ - "01-seedtts-01_promptvn.wav", - null, - null - ] - }, - { - "id": 16, - "type": "LoadImage", - "pos": [ - 20189.94934337644, - -137.90936898810804 - ], - "size": [ - 270, - 314 - ], - "flags": {}, - "order": 7, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 22 - ] - }, - { - "name": "MASK", - "type": "MASK", - "links": [ - 23 - ] - } - ], - "properties": { - "Node name for S&R": "LoadImage", - "image": "clipspace/clipspace-painted-masked-1770205344996.png [input]" - }, - "widgets_values": [ - "clipspace/clipspace-painted-masked-1770205344996.png [input]", - "image" - ] - }, - { - "id": 22, - "type": "Note", - "pos": [ - 23356.144432636724, - -881.6479805868449 - ], - "size": [ - 390.28170729224803, - 246.35694222626103 - ], - "flags": {}, - "order": 8, - "mode": 0, - "inputs": [], - "outputs": [], - "properties": {}, - "widgets_values": [ - "mode=a2mv\ntransformer_path=interact-avatar\ntest_data_path=InterDemo/TIA2MV/demo_tia2mv.json\nback_append_frame=1\nframe_num=101\n\n# for pose dirven with audio\nmode=ap2v\ntransformer_path=interact-avatar\ntest_data_path=InterDemo/TIAP2V/demo_ap2v.json\nback_append_frame=1\nframe_num=133\n\nfor song with action\n# mode=a2mv\n# transformer_path=interact-avatar-long\n# test_data_path=demo_song_action.json\n# back_append_frame=2\n# frame_num=1000" - ], - "color": "#432", - "bgcolor": "#653" - }, { "id": 17, "type": "VAEDecode", @@ -551,8 +56,8 @@ 46 ], "flags": {}, - "order": 24, - "mode": 0, + "order": 37, + "mode": 2, "inputs": [ { "name": "samples", @@ -579,41 +84,6 @@ }, "widgets_values": [] }, - { - "id": 30, - "type": "InteractAvatar_SM_Model", - "pos": [ - 22978.61090737092, - -788.1824536240806 - ], - "size": [ - 278.662109375, - 154 - ], - "flags": {}, - "order": 9, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "model", - "type": "MODEL", - "links": [ - 38 - ] - } - ], - "properties": { - "Node name for S&R": "InteractAvatar_SM_Model" - }, - "widgets_values": [ - "interact-avatar.safetensors", - "none", - "none", - 1, - true - ] - }, { "id": 23, "type": "CLIPLoader", @@ -626,8 +96,8 @@ 106 ], "flags": {}, - "order": 10, - "mode": 0, + "order": 0, + "mode": 2, "inputs": [], "outputs": [ { @@ -659,8 +129,8 @@ 58 ], "flags": {}, - "order": 11, - "mode": 0, + "order": 1, + "mode": 2, "inputs": [], "outputs": [ { @@ -679,33 +149,6 @@ "Wan2.2_VAE.pth" ] }, - { - "id": 31, - "type": "SaveImage", - "pos": [ - 23358.789761969365, - -99.02376747958866 - ], - "size": [ - 270, - 270 - ], - "flags": {}, - "order": 17, - "mode": 0, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 40 - } - ], - "outputs": [], - "properties": {}, - "widgets_values": [ - "ComfyUI" - ] - }, { "id": 32, "type": "InvertMask", @@ -718,8 +161,8 @@ 26 ], "flags": {}, - "order": 18, - "mode": 0, + "order": 26, + "mode": 2, "inputs": [ { "name": "mask", @@ -753,8 +196,8 @@ 314.00000000000006 ], "flags": {}, - "order": 12, - "mode": 0, + "order": 2, + "mode": 2, "inputs": [], "outputs": [ { @@ -779,19 +222,598 @@ ] }, { - "id": 26, + "id": 21, + "type": "InteractAvatar_SM_Sampler", + "pos": [ + 23343.637071139194, + -502.7631653504566 + ], + "size": [ + 281.2984375, + 318 + ], + "flags": {}, + "order": 34, + "mode": 2, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 38 + }, + { + "name": "data_dict", + "type": "CONDITIONING", + "link": 51 + } + ], + "outputs": [ + { + "name": "Latent", + "type": "LATENT", + "links": [ + 25 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Sampler" + }, + "widgets_values": [ + 20, + 381967489, + "randomize", + 5, + 5, + 7.5, + 800, + 25, + true, + false + ] + }, + { + "id": 18, + "type": "CreateVideo", + "pos": [ + 23633.536530795816, + -481.4985380777255 + ], + "size": [ + 270, + 78 + ], + "flags": {}, + "order": 40, + "mode": 2, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 27 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 39 + } + ], + "outputs": [ + { + "name": "VIDEO", + "type": "VIDEO", + "links": [ + 28 + ] + } + ], + "properties": { + "Node name for S&R": "CreateVideo" + }, + "widgets_values": [ + 25 + ] + }, + { + "id": 34, + "type": "InteractAvatar_SM_Predata", + "pos": [ + 20575.27704051971, + -562.9400484305286 + ], + "size": [ + 430.9281606020304, + 484 + ], + "flags": {}, + "order": 25, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 52 + }, + { + "name": "vae", + "type": "VAE", + "link": 53 + }, + { + "name": "images", + "type": "IMAGE", + "link": 58 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 57 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 56 + }, + { + "name": "object_images", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "object_mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "name": "data_dict", + "type": "CONDITIONING", + "links": [ + 54 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Predata" + }, + "widgets_values": [ + 512, + "Hand clenches in a fist to cheer,\nRest chin on one hand in thought.", + "bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards", + "(Hand clenches in a fist) (Cheer with clenched fist),\n(Raise one hand slowly to the chin) (Rest chin on the hand)", + 1000, + "a2mv", + 2, + true, + "", + "" + ] + }, + { + "id": 35, + "type": "InteractAvatar_SM_Model", + "pos": [ + 20699.49201929431, + -787.3366582507534 + ], + "size": [ + 278.662109375, + 154 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 55 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Model" + }, + "widgets_values": [ + "interact-avatar-long.safetensors", + "none", + "none", + 2, + true + ] + }, + { + "id": 8, + "type": "CLIPLoader", + "pos": [ + 20233.043156660322, + -827.0414995742086 + ], + "size": [ + 270, + 106 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 52 + ] + } + ], + "properties": { + "Node name for S&R": "CLIPLoader" + }, + "widgets_values": [ + "umt5_xxl_fp8_e4m3fn_scaled.safetensors", + "wan", + "default" + ] + }, + { + "id": 5, + "type": "VAELoader", + "pos": [ + 20233.964106665513, + -666.2111245505673 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 4, + 53 + ] + } + ], + "properties": { + "Node name for S&R": "VAELoader" + }, + "widgets_values": [ + "Wan2.2_VAE.pth" + ] + }, + { + "id": 13, + "type": "InteractAvatar_SM_Sampler", + "pos": [ + 21082.87523550645, + -591.8482037787247 + ], + "size": [ + 281.2984375, + 318 + ], + "flags": {}, + "order": 30, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 55 + }, + { + "name": "data_dict", + "type": "CONDITIONING", + "link": 54 + } + ], + "outputs": [ + { + "name": "Latent", + "type": "LATENT", + "links": [ + 15 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Sampler" + }, + "widgets_values": [ + 20, + 1034563315, + "randomize", + 5, + 5, + 7.5, + 800, + 25, + 1000, + false + ] + }, + { + "id": 9, "type": "LoadImage", "pos": [ - 22246.524498547184, - -338.06913418740334 + 19817.499679613604, + -756.4208862352102 + ], + "size": [ + 270, + 314 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 58 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "001.png", + "image" + ] + }, + { + "id": 10, + "type": "LoadImage", + "pos": [ + 19839.92752091642, + -392.48311359683214 + ], + "size": [ + 270, + 314 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 57 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "001 (1).png", + "image" + ] + }, + { + "id": 12, + "type": "LoadAudio", + "pos": [ + 20226.379812505147, + -336.1859815149922 + ], + "size": [ + 285.46652238349634, + 221.47288685617252 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 56, + 59 + ] + } + ], + "properties": { + "Node name for S&R": "LoadAudio" + }, + "widgets_values": [ + "002.WAV", + null, + null + ] + }, + { + "id": 6, + "type": "CreateVideo", + "pos": [ + 21637.779478502343, + -699.0965481869938 + ], + "size": [ + 270, + 78 + ], + "flags": {}, + "order": 36, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 5 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 59 + } + ], + "outputs": [ + { + "name": "VIDEO", + "type": "VIDEO", + "links": [ + 6 + ] + } + ], + "properties": { + "Node name for S&R": "CreateVideo" + }, + "widgets_values": [ + 25 + ] + }, + { + "id": 30, + "type": "InteractAvatar_SM_Model", + "pos": [ + 22978.61090737092, + -788.1824536240806 + ], + "size": [ + 278.662109375, + 154 + ], + "flags": {}, + "order": 9, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 38 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Model" + }, + "widgets_values": [ + "interact-avatar.safetensors", + "none", + "none", + 1, + true + ] + }, + { + "id": 19, + "type": "SaveVideo", + "pos": [ + 23885.79947024992, + -672.7530469258098 + ], + "size": [ + 478, + 815 + ], + "flags": {}, + "order": 42, + "mode": 2, + "inputs": [ + { + "name": "video", + "type": "VIDEO", + "link": 28 + } + ], + "outputs": [], + "properties": {}, + "widgets_values": [ + "video/ComfyUI", + "auto", + "auto" + ] + }, + { + "id": 29, + "type": "LoadImage", + "pos": [ + 22514.549266427664, + -159.4141741239947 ], "size": [ 270, 314.0000000000001 ], "flags": {}, - "order": 13, - "mode": 0, + "order": 10, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 49 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": [ + 41 + ] + } + ], + "properties": { + "Node name for S&R": "LoadImage", + "image": "clipspace/clipspace-painted-masked-1770619718083.png [input]" + }, + "widgets_values": [ + "clipspace/clipspace-painted-masked-1770619718083.png [input]", + "image" + ] + }, + { + "id": 26, + "type": "LoadImage", + "pos": [ + 22198.935219632036, + -241.44847699606 + ], + "size": [ + 270, + 314.0000000000001 + ], + "flags": {}, + "order": 11, + "mode": 2, "inputs": [], "outputs": [ { @@ -816,46 +838,676 @@ ] }, { - "id": 29, + "id": 28, + "type": "LoadAudio", + "pos": [ + 22563.286878037456, + -454.15325713544894 + ], + "size": [ + 270, + 136 + ], + "flags": {}, + "order": 12, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 39, + 46 + ] + } + ], + "properties": { + "Node name for S&R": "LoadAudio" + }, + "widgets_values": [ + "02-seedtts-02_promptvn.wav", + null, + null + ] + }, + { + "id": 37, + "type": "VAEDecode", + "pos": [ + 23746.42226156669, + 683.8375855155487 + ], + "size": [ + 140, + 46 + ], + "flags": {}, + "order": 38, + "mode": 2, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 62 + }, + { + "name": "vae", + "type": "VAE", + "link": 63 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 67 + ] + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + }, + "widgets_values": [] + }, + { + "id": 38, + "type": "CLIPLoader", + "pos": [ + 22596.091780321924, + 513.7922930870564 + ], + "size": [ + 270, + 106 + ], + "flags": {}, + "order": 13, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 81 + ] + } + ], + "properties": { + "Node name for S&R": "CLIPLoader" + }, + "widgets_values": [ + "umt5_xxl_fp8_e4m3fn_scaled.safetensors", + "wan", + "default" + ] + }, + { + "id": 39, + "type": "VAELoader", + "pos": [ + 22606.330577438424, + 681.6652269738995 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 14, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 63, + 82 + ] + } + ], + "properties": { + "Node name for S&R": "VAELoader" + }, + "widgets_values": [ + "Wan2.2_VAE.pth" + ] + }, + { + "id": 40, + "type": "InvertMask", + "pos": [ + 22930.96984722589, + 1384.1367731799974 + ], + "size": [ + 140, + 26 + ], + "flags": {}, + "order": 28, + "mode": 2, + "inputs": [ + { + "name": "mask", + "type": "MASK", + "link": 64 + } + ], + "outputs": [ + { + "name": "MASK", + "type": "MASK", + "links": [ + 87 + ] + } + ], + "properties": { + "Node name for S&R": "InvertMask" + }, + "widgets_values": [] + }, + { + "id": 42, + "type": "InteractAvatar_SM_Sampler", + "pos": [ + 23367.805629316197, + 812.1519245324877 + ], + "size": [ + 281.2984375, + 318 + ], + "flags": {}, + "order": 35, + "mode": 2, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 65 + }, + { + "name": "data_dict", + "type": "CONDITIONING", + "link": 66 + } + ], + "outputs": [ + { + "name": "Latent", + "type": "LATENT", + "links": [ + 62 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Sampler" + }, + "widgets_values": [ + 20, + 2144302668, + "randomize", + 5, + 5, + 7.5, + 800, + 25, + true, + false + ] + }, + { + "id": 43, + "type": "CreateVideo", + "pos": [ + 23657.70508897282, + 833.4165518052188 + ], + "size": [ + 270, + 78 + ], + "flags": {}, + "order": 41, + "mode": 2, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 67 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 68 + } + ], + "outputs": [ + { + "name": "VIDEO", + "type": "VIDEO", + "links": [ + 79 + ] + } + ], + "properties": { + "Node name for S&R": "CreateVideo" + }, + "widgets_values": [ + 25 + ] + }, + { + "id": 56, + "type": "InteractAvatar_SM_Model", + "pos": [ + 23002.779465547923, + 526.732636258864 + ], + "size": [ + 278.662109375, + 154 + ], + "flags": {}, + "order": 15, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 65 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Model" + }, + "widgets_values": [ + "interact-avatar.safetensors", + "none", + "none", + 1, + true + ] + }, + { + "id": 57, + "type": "SaveVideo", + "pos": [ + 23909.968028426923, + 642.1620429571349 + ], + "size": [ + 478, + 815 + ], + "flags": {}, + "order": 43, + "mode": 2, + "inputs": [ + { + "name": "video", + "type": "VIDEO", + "link": 79 + } + ], + "outputs": [], + "properties": {}, + "widgets_values": [ + "video/ComfyUI", + "auto", + "auto" + ] + }, + { + "id": 61, + "type": "SaveImage", + "pos": [ + 23393.684679030324, + 1203.97314586563 + ], + "size": [ + 270, + 270 + ], + "flags": {}, + "order": 27, + "mode": 2, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 80 + } + ], + "outputs": [], + "properties": {}, + "widgets_values": [ + "ComfyUI" + ] + }, + { + "id": 60, + "type": "LoadAudio", + "pos": [ + 22594.606342137093, + 921.544533089897 + ], + "size": [ + 309.32998257449435, + 155.0690824603613 + ], + "flags": {}, + "order": 16, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 68, + 85 + ] + } + ], + "properties": { + "Node name for S&R": "LoadAudio" + }, + "widgets_values": [ + "bzya5p228n0xlz96_clip_0000_118.wav", + null, + null + ] + }, + { + "id": 65, + "type": "GetVideoComponents", + "pos": [ + 22618.49777155802, + 791.805889808121 + ], + "size": [ + 140, + 66 + ], + "flags": {}, + "order": 29, + "mode": 2, + "inputs": [ + { + "name": "video", + "type": "VIDEO", + "link": 89 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 90 + ] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null + }, + { + "name": "fps", + "type": "FLOAT", + "links": null + } + ], + "properties": { + "Node name for S&R": "GetVideoComponents" + }, + "widgets_values": [] + }, + { + "id": 58, "type": "LoadImage", "pos": [ - 22582.48287269269, - -271.44503357861754 + 22612.610519138565, + 1175.7618158730838 ], "size": [ 270, 314.0000000000001 ], "flags": {}, - "order": 14, - "mode": 0, + "order": 17, + "mode": 2, "inputs": [], "outputs": [ { "name": "IMAGE", "type": "IMAGE", "links": [ - 40, - 49 + 80, + 86 ] }, { "name": "MASK", "type": "MASK", "links": [ - 41 + 64 ] } ], "properties": { "Node name for S&R": "LoadImage", - "image": "clipspace/clipspace-painted-masked-1770263662180.png [input]" + "image": "clipspace/clipspace-painted-masked-1770639930165.png [input]" }, "widgets_values": [ - "clipspace/clipspace-painted-masked-1770263662180.png [input]", + "clipspace/clipspace-painted-masked-1770639930165.png [input]", "image" ] }, + { + "id": 64, + "type": "LoadVideo", + "pos": [ + 22089.330733282997, + 636.8695948176855 + ], + "size": [ + 501.52448895366615, + 360.5854222592378 + ], + "flags": {}, + "order": 18, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "VIDEO", + "type": "VIDEO", + "links": [ + 89 + ] + } + ], + "properties": { + "Node name for S&R": "LoadVideo" + }, + "widgets_values": [ + "self_collect_20250508_trans_slpit_self_collect_20250508_24_000_crop_0125.mp4", + "image" + ] + }, + { + "id": 68, + "type": "VHS_LoadVideo", + "pos": [ + 22198.183836371754, + 1011.1363854893792 + ], + "size": [ + 261.6533203125, + 462.0876116071429 + ], + "flags": {}, + "order": 19, + "mode": 2, + "inputs": [ + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 93 + ] + }, + { + "name": "frame_count", + "type": "INT", + "links": null + }, + { + "name": "audio", + "type": "AUDIO", + "links": null + }, + { + "name": "video_info", + "type": "VHS_VIDEOINFO", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_LoadVideo" + }, + "widgets_values": { + "video": "pose1 (2).mp4", + "force_rate": 0, + "custom_width": 0, + "custom_height": 0, + "frame_load_cap": 0, + "skip_first_frames": 0, + "select_every_nth": 1, + "format": "AnimateDiff", + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "pose1 (2).mp4", + "type": "input", + "format": "video/mp4", + "force_rate": 0, + "custom_width": 0, + "custom_height": 0, + "frame_load_cap": 0, + "skip_first_frames": 0, + "select_every_nth": 1 + } + } + } + }, + { + "id": 62, + "type": "InteractAvatar_SM_Predata", + "pos": [ + 22931.66087384489, + 755.8245633100523 + ], + "size": [ + 398.2881522781572, + 512.4055027016293 + ], + "flags": {}, + "order": 32, + "mode": 2, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 81 + }, + { + "name": "vae", + "type": "VAE", + "link": 82 + }, + { + "name": "images", + "type": "IMAGE", + "link": 90 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 93 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 85 + }, + { + "name": "object_images", + "shape": 7, + "type": "IMAGE", + "link": 86 + }, + { + "name": "object_mask", + "shape": 7, + "type": "MASK", + "link": 87 + } + ], + "outputs": [ + { + "name": "data_dict", + "type": "CONDITIONING", + "links": [ + 66 + ] + } + ], + "properties": { + "Node name for S&R": "InteractAvatar_SM_Predata" + }, + "widgets_values": [ + 512, + "A woman with long dark hair, wearing large gold hoop earrings and a navy blue hoodie, sits holding a thin, flat, black rectangular object with vertical ridges. She is indoors in a room with white walls, a wooden cabinet visible behind her, and several cardboard boxes on the floor around her.", + "bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards", + "", + 133, + "ap2v", + 1, + "", + "", + "" + ] + }, { "id": 33, "type": "InteractAvatar_SM_Predata", @@ -864,12 +1516,12 @@ -559.0905265728924 ], "size": [ - 417.9086048651043, - 534.3207101901768 + 418.76042765551756, + 532.3827305146999 ], "flags": {}, - "order": 20, - "mode": 0, + "order": 31, + "mode": 2, "inputs": [ { "name": "clip", @@ -936,152 +1588,116 @@ ] }, { - "id": 28, - "type": "LoadAudio", + "id": 55, + "type": "Note", "pos": [ - 22564.72897739852, - -467.13215138503244 + 23388.178987328633, + 448.9991023258981 ], "size": [ - 270, - 136 + 390.28170729224803, + 246.35694222626103 ], "flags": {}, - "order": 15, - "mode": 0, + "order": 20, + "mode": 2, "inputs": [], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 39, - 46 - ] - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, + "outputs": [], + "properties": {}, "widgets_values": [ - "02-seedtts-02_promptvn.wav", - null, - null - ] + "\n音频及pose驱动视频\n# for pose dirven with audio\nmode=ap2v\ntransformer_path=interact-avatar\ntest_data_path=InterDemo/TIAP2V/demo_ap2v.json\nback_append_frame=1\nframe_num=133\n提示词格式须遵循json文件内的书写要求\n如果是中文,加单引号括起来,圆括号不能删除,是格式要求" + ], + "color": "#432", + "bgcolor": "#653" }, { - "id": 21, - "type": "InteractAvatar_SM_Sampler", + "id": 69, + "type": "Note", "pos": [ - 23343.637071139194, - -502.7631653504566 + 22227.41603447046, + 506.35755016824277 ], "size": [ - 281.2984375, - 318 + 328.2714347664587, + 88 + ], + "flags": {}, + "order": 21, + "mode": 2, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "垫图可以输入单张图片\ncan input 1 image" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 22, + "type": "Note", + "pos": [ + 23364.01042915163, + -801.5578342533277 + ], + "size": [ + 369.44056838048346, + 182.10534512148888 ], "flags": {}, "order": 22, - "mode": 0, - "inputs": [ - { - "name": "model", - "type": "MODEL", - "link": 38 - }, - { - "name": "data_dict", - "type": "CONDITIONING", - "link": 51 - } - ], - "outputs": [ - { - "name": "Latent", - "type": "LATENT", - "links": [ - 25 - ] - } - ], - "properties": { - "Node name for S&R": "InteractAvatar_SM_Sampler" - }, + "mode": 2, + "inputs": [], + "outputs": [], + "properties": {}, "widgets_values": [ - 20, - 1714210267, - "randomize", - 5, - 5, - 7.5, - 800, - 25, - true, - false - ] + "音频及文本驱动视频\nmode=a2mv\ntransformer_path=interact-avatar\ntest_data_path=InterDemo/TIA2MV/demo_tia2mv.json\nback_append_frame=1\nframe_num=101\n提示词格式须遵循json文件内的书写要求\n如果是中文,加单引号括起来,圆括号不能删除,是格式要求\n这里最下面的提示词不用分成多行\n" + ], + "color": "#432", + "bgcolor": "#653" }, { - "id": 18, - "type": "CreateVideo", + "id": 14, + "type": "Note", "pos": [ - 23633.536530795816, - -481.4985380777255 + 21072.413020403983, + -891.9647068066879 ], "size": [ - 270, - 78 + 374.54971426245174, + 148.03198579002412 ], "flags": {}, - "order": 26, + "order": 23, "mode": 0, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 27 - }, - { - "name": "audio", - "shape": 7, - "type": "AUDIO", - "link": 39 - } - ], - "outputs": [ - { - "name": "VIDEO", - "type": "VIDEO", - "links": [ - 28 - ] - } - ], - "properties": { - "Node name for S&R": "CreateVideo" - }, + "inputs": [], + "outputs": [], + "properties": {}, "widgets_values": [ - 25 - ] + "# 音频及动作文本驱动\nfor song with action\n# mode=a2mv\n# transformer_path=interact-avatar-long\n# test_data_path=demo_song_action.json\n# back_append_frame=2\n# frame_num=1000\n提示词格式须遵循json文件内的书写要求,最上面的提示词是要分行的,最下面的也一样,必须行数相同,内容根据你的要求书写,\n如果是中文,加单引号括起来,圆括号不能删除,是格式要求" + ], + "color": "#432", + "bgcolor": "#653" }, { - "id": 19, + "id": 7, "type": "SaveVideo", "pos": [ - 23885.79947024992, - -672.7530469258098 + 21459.000968860535, + -517.537289595406 ], "size": [ - 478, - 815 + 490.9504670058959, + 620.7824650219009 ], "flags": {}, - "order": 27, + "order": 39, "mode": 0, "inputs": [ { "name": "video", "type": "VIDEO", - "link": 28 + "link": 6 } ], "outputs": [], @@ -1091,6 +1707,29 @@ "auto", "auto" ] + }, + { + "id": 70, + "type": "Note", + "pos": [ + 21031.528040733716, + -210.4890065253783 + ], + "size": [ + 377.1717131007499, + 88 + ], + "flags": {}, + "order": 24, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "提示词要求,最上面的提示词对应对应最下面的,行数须相同,可以加词或者行数" + ], + "color": "#432", + "bgcolor": "#653" } ], "links": [ @@ -1118,14 +1757,6 @@ 0, "VIDEO" ], - [ - 13, - 1, - 0, - 13, - 0, - "MODEL" - ], [ 15, 13, @@ -1134,70 +1765,6 @@ 0, "LATENT" ], - [ - 16, - 8, - 0, - 15, - 0, - "CLIP" - ], - [ - 17, - 5, - 0, - 15, - 1, - "VAE" - ], - [ - 18, - 15, - 0, - 13, - 1, - "CONDITIONING" - ], - [ - 19, - 9, - 0, - 15, - 2, - "IMAGE" - ], - [ - 20, - 10, - 0, - 15, - 3, - "IMAGE" - ], - [ - 22, - 16, - 0, - 15, - 5, - "IMAGE" - ], - [ - 23, - 16, - 1, - 15, - 6, - "MASK" - ], - [ - 24, - 12, - 0, - 15, - 4, - "AUDIO" - ], [ 25, 21, @@ -1246,14 +1813,6 @@ 1, "AUDIO" ], - [ - 40, - 29, - 0, - 31, - 0, - "IMAGE" - ], [ 41, 29, @@ -1325,6 +1884,206 @@ 21, 1, "CONDITIONING" + ], + [ + 52, + 8, + 0, + 34, + 0, + "CLIP" + ], + [ + 53, + 5, + 0, + 34, + 1, + "VAE" + ], + [ + 54, + 34, + 0, + 13, + 1, + "CONDITIONING" + ], + [ + 55, + 35, + 0, + 13, + 0, + "MODEL" + ], + [ + 56, + 12, + 0, + 34, + 4, + "AUDIO" + ], + [ + 57, + 10, + 0, + 34, + 3, + "IMAGE" + ], + [ + 58, + 9, + 0, + 34, + 2, + "IMAGE" + ], + [ + 59, + 12, + 0, + 6, + 1, + "AUDIO" + ], + [ + 62, + 42, + 0, + 37, + 0, + "LATENT" + ], + [ + 63, + 39, + 0, + 37, + 1, + "VAE" + ], + [ + 64, + 58, + 1, + 40, + 0, + "MASK" + ], + [ + 65, + 56, + 0, + 42, + 0, + "MODEL" + ], + [ + 66, + 62, + 0, + 42, + 1, + "CONDITIONING" + ], + [ + 67, + 37, + 0, + 43, + 0, + "IMAGE" + ], + [ + 68, + 60, + 0, + 43, + 1, + "AUDIO" + ], + [ + 79, + 43, + 0, + 57, + 0, + "VIDEO" + ], + [ + 80, + 58, + 0, + 61, + 0, + "IMAGE" + ], + [ + 81, + 38, + 0, + 62, + 0, + "CLIP" + ], + [ + 82, + 39, + 0, + 62, + 1, + "VAE" + ], + [ + 85, + 60, + 0, + 62, + 4, + "AUDIO" + ], + [ + 86, + 58, + 0, + 62, + 5, + "IMAGE" + ], + [ + 87, + 40, + 0, + 62, + 6, + "MASK" + ], + [ + 89, + 64, + 0, + 65, + 0, + "VIDEO" + ], + [ + 90, + 65, + 0, + 62, + 2, + "IMAGE" + ], + [ + 93, + 68, + 0, + 62, + 3, + "IMAGE" ] ], "groups": [ @@ -1353,16 +2112,29 @@ "color": "#3f789e", "font_size": 24, "flags": {} + }, + { + "id": 4, + "title": "Group", + "bounding": [ + 22102.377331939046, + 359.951986141428, + 2291.6857937761633, + 1151.0443112641383 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} } ], "config": {}, "extra": { "workflowRendererVersion": "LG", "ds": { - "scale": 0.6934334949442076, + "scale": 0.5730855330117559, "offset": [ - -21975.66153358101, - 1046.1905376736581 + -19679.843308053405, + 1135.4929857182988 ] }, "frontendVersion": "1.37.11", diff --git a/example_workflows/example_ap2v.png b/example_workflows/example_ap2v.png new file mode 100644 index 0000000..889a416 Binary files /dev/null and b/example_workflows/example_ap2v.png differ diff --git a/model_loader_utils.py b/model_loader_utils.py index 3f9f187..7622e65 100644 --- a/model_loader_utils.py +++ b/model_loader_utils.py @@ -13,7 +13,7 @@ import torchaudio import folder_paths from comfy.utils import common_upscale,ProgressBar from safetensors.torch import load_file - +import soundfile as sf import comfy.model_management as mm from pathlib import PureWindowsPath cur_path = os.path.dirname(os.path.abspath(__file__)) @@ -136,7 +136,22 @@ def clear_comfyui_cache(): max_gpu_memory = torch.cuda.max_memory_allocated() print(f"After Max GPU memory allocated: {max_gpu_memory / 1000 ** 3:.2f} GB") +# def trans2path(audio): +# if audio is None: +# return None +# import io as io_base +# audio_file_prefix = ''.join(random.choice("0123456789") for _ in range(6)) +# audio_file = os.path.join(folder_paths.get_input_directory(), f"audio_{audio_file_prefix}_temp.wav") +# buff = io_base.BytesIO() + +# torchaudio.save(buff, audio["waveform"].squeeze(0), audio["sample_rate"], format="FLAC") +# with open(audio_file, 'wb') as f: +# f.write(buff.getbuffer()) +# return audio_file def trans2path(audio): + """ + 修正版:使用 soundfile 代替 torchaudio.save 以避开 torchcodec 的环境报错。 + """ if audio is None: return None import io as io_base @@ -144,12 +159,27 @@ def trans2path(audio): audio_file = os.path.join(folder_paths.get_input_directory(), f"audio_{audio_file_prefix}_temp.wav") buff = io_base.BytesIO() - torchaudio.save(buff, audio["waveform"].squeeze(0), audio["sample_rate"], format="FLAC") + # --- 修正逻辑开始 --- + # ComfyUI 音频格式通常为 [Batch, Channels, Samples] -> [1, C, S] + # 我们需要将其转换为 NumPy,并调整维度为 soundfile 要求的 [Samples, Channels] + waveform = audio["waveform"].squeeze(0).cpu().numpy() # 结果为 [C, S] + sample_rate = audio["sample_rate"] + + if waveform.ndim == 2: + waveform = waveform.T # 转置为 [S, C] + + # 使用 soundfile 直接写入内存流,不触发 torchaudio 的后端检测 + sf.write(buff, waveform, sample_rate, format="FLAC") + # --- 修正逻辑结束 --- + with open(audio_file, 'wb') as f: f.write(buff.getbuffer()) return audio_file + + + def encode_image( image, vae): if image is None: return None diff --git a/test_wanx_tia2mv_obj_back.py b/test_wanx_tia2mv_obj_back.py index 53b2cbb..423cf49 100644 --- a/test_wanx_tia2mv_obj_back.py +++ b/test_wanx_tia2mv_obj_back.py @@ -269,7 +269,6 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode dw_seqs = [dw.resize(dw_img.size, Image.LANCZOS) for dw in dw_seqs] dwpose_len = len(dw_seqs) dwpose_frame_num = (dwpose_len - 1) // 4 * 4 + 1 - # pre audio if audio_path is not None: audio_input, sampling_rate = librosa.load(audio_path, sr=16000) @@ -281,7 +280,7 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode audio_frame_num = dwpose_frame_num sampling_rate = 16000 - if dw_seqs is not None: + if dw_seqs is not None: # 对齐pose和音频帧数 dw_seqs = dw_seqs[:frame_num] if audio_frame_num > dwpose_frame_num: padding_dwpose = dw_seqs[-1] @@ -292,9 +291,10 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode audio_frames_clip = np.concatenate([audio_frames_clip, padding_audio], axis=0) audio_frame_num = dwpose_frame_num if dw_seqs is None and mode == 'ap2v': - dw_seqs = [dw_img] * audio_frame_num + dw_seqs = [dw_img] * audio_frame_num #对齐帧数 + + frame_num = min(frame_num,min(audio_frame_num,dwpose_frame_num)) #对齐推理帧和音频帧数 - frame_num = min(frame_num,min(audio_frame_num,dwpose_frame_num)) audio_frames_clip = audio_frames_clip[:int(frame_num * sampling_rate / 25)] wav2vec_feature_extractor, audio_encoder= custom_init(device, wav2vec_dir) @@ -330,8 +330,8 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode if mode in ['a2v','a2mv','mv','i2v']: dw_seqs = None - vae_stride=WAN_CONFIGS['ti2v-5B'].vae_stride - patch_size=WAN_CONFIGS['ti2v-5B'].patch_size + vae_stride=WAN_CONFIGS['ti2v-5B'].vae_stride #(4, 16, 16) + patch_size=WAN_CONFIGS['ti2v-5B'].patch_size #(1, 2, 2) sp_size=1 if back_append_frame==1: @@ -345,15 +345,16 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode if dw_seqs is not None: if isinstance(dw_seqs, list): # If input is a list of PIL Images - processed_poses = [phi2narry(p) for p in dw_seqs] - cond_pose_sequence = torch.stack(processed_poses).to(device) + processed_poses = [phi2narry(p.convert('RGB')) for p in dw_seqs] + cond_pose_sequence=torch.cat(processed_poses).to(device) + #cond_pose_sequence = torch.stack(processed_poses).to(device) else: # If input is already a tensor cond_pose_sequence = dw_seqs.to(device) dwpose_len = cond_pose_sequence.shape[0] dwpose_len = (dwpose_len - 1) // vae_stride[0] * vae_stride[0] + 1 frame_num = min(frame_num, dwpose_len) cond_pose_sequence = cond_pose_sequence[:frame_num] - + #print(cond_pose_sequence.shape) #torch.Size([133, 256, 448, 3]) else: cond_pose_sequence = torch.ones((frame_num, pose_ref_img.shape[1], pose_ref_img.shape[2], pose_ref_img.shape[3]), device=pose_ref_img.device, dtype=pose_ref_img.dtype) * 0.5 @@ -419,6 +420,7 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode mode=mode, frame_num=frame_num, max_frames_num=frame_num, + short_side=short_side, ) else: curr_cond_image=phi2narry(img) #BHWC @@ -434,8 +436,8 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode # Pose Sequence 处理 if dw_seqs is not None: if isinstance(dw_seqs, list): - processed_poses = [phi2narry(p) for p in dw_seqs] - cond_pose_sequence_full = torch.stack(processed_poses).to(device) + processed_poses = [phi2narry(p.convert('RGB')) for p in dw_seqs] + cond_pose_sequence=torch.cat(processed_poses).to(device) else: cond_pose_sequence_full = dw_seqs.to(device) else: @@ -535,5 +537,6 @@ def perdata( clip,vae,images,dw_iamges,object_images,object_mask,audio_path,mode frame_num=frame_num, vae=vae, max_frames_num=frame_num, + short_side=short_side, ) return data_dict diff --git a/wan/modules/model_tia2mv_rope_back.py b/wan/modules/model_tia2mv_rope_back.py index f6a7a11..2be13ae 100644 --- a/wan/modules/model_tia2mv_rope_back.py +++ b/wan/modules/model_tia2mv_rope_back.py @@ -1479,6 +1479,7 @@ class WanModel(ModelMixin, ConfigMixin): use_gradient_checkpointing_offload=False, cond_flag=False, gpu_manager=None, + up_scale=2.75, **kwargs ): r""" @@ -1512,16 +1513,14 @@ class WanModel(ModelMixin, ConfigMixin): device = self.patch_embedding.weight.device if self.freqs.device != device: self.freqs = self.freqs.to(device) - # print(x.shape) - # embeddings if x is not None: x = [self.patch_embedding(u.unsqueeze(0)) for u in x] # [b, 1, dim, t, h/2, w/2] -> [b,seq_len,dim 1536] grid_sizes = torch.stack( [torch.tensor(u.shape[2:], dtype=torch.long) for u in x]) # [1, 3], thw frame_l = x[0].shape[-1] * x[0].shape[-2] x = [u.flatten(2).transpose(1, 2) for u in x] # [b, dim, thw/4] => [b, thw/4, dim] - seq_lens = torch.tensor([u.size(1) for u in x], dtype=torch.long) - + seq_lens = torch.tensor([u.size(1) for u in x], dtype=torch.long) + #print(seq_lens,"seq lens") #tensor([10368]) seq lens if seq_len!=0: assert seq_lens.max() <= seq_len x = torch.cat([ @@ -1672,7 +1671,6 @@ class WanModel(ModelMixin, ConfigMixin): ) iii = 0 pre_motion = torch.zeros_like(motion, requires_grad=True) - #print(f"len(self.blocks): {len(self.blocks)}",len(self.zero_motion_proj_blocks),len(self.motion_blocks)) #30,30,30 for idx,block in enumerate(self.blocks): if gpu_manager is not None: if idx < len(self.blocks): @@ -1690,11 +1688,12 @@ class WanModel(ModelMixin, ConfigMixin): residual_motion = residual_motion.transpose(1, 2).reshape(bb, -1, ff, hh, ww) residual_motion = residual_motion.permute(0,2,1,3,4).reshape(bb*ff, -1, hh, ww) residual_motion_up = F.interpolate( - residual_motion, scale_factor=(2.75, 2.75), + residual_motion, scale_factor=(up_scale, up_scale), #2.75 = 704/256,2.0=512/256 mode='bilinear', align_corners=False ).to(motion.dtype) H_out, W_out = residual_motion_up.shape[-2], residual_motion_up.shape[-1] last_part_len = H_out * W_out + residual_motion_up = residual_motion_up.reshape(bb, ff, -1, H_out, W_out).permute(0,2,1,3,4) if gpu_manager is not None: @@ -1706,29 +1705,14 @@ class WanModel(ModelMixin, ConfigMixin): prev_module_ = gpu_manager.managed_motion_proj_modules[idx - 1] if hasattr(prev_module_, 'to'): prev_module_.to('cpu') - + value_to_add = self.zero_motion_proj_blocks[idx](residual_motion_up.flatten(2).transpose(1, 2)) - #print(f"value_to_add shape: {value_to_add.shape}",f"x shape: {x.shape}",f"last_part_len: {last_part_len}") #value_to_add shape: torch.Size([1, 15972, 3072]) + #print(f"value_to_add shape: {value_to_add.shape}",f"x shape: {x.shape}",f"last_part_len: {last_part_len}") + #value_to_add shape: torch.Size([1, 19602, 3072]) x shape: torch.Size([1, 10368, 3072]) last_part_len: 726 # if use short_side=512 got error when upscale=2.75 + + x[:, :-last_part_len, :] = x[:, :-last_part_len, :] + value_to_add[:, :-last_part_len, :] - #x[:, :-last_part_len, :] = x[:, :-last_part_len, :] + value_to_add[:, :-last_part_len, :] - - # 确保 last_part_len 不超过 x 的序列长度 - last_part_len = min(last_part_len, x.size(1)) - - # 计算 x 中需要更新的长度 - x_update_len = x.size(1) - last_part_len - - # 确保 value_to_add 的长度与 x_update_len 匹配 - if value_to_add.size(1) >= x_update_len: - # 如果 value_to_add 足够长,只使用前 x_update_len 个位置 - x[:, :x_update_len, :] = x[:, :x_update_len, :] + value_to_add[:, :x_update_len, :] - else: - # 如果 value_to_add 比 x_update_len 短,使用全部 value_to_add - x[:, :value_to_add.size(1), :] = x[:, :value_to_add.size(1), :] + value_to_add - - ## apply motion block - if gpu_manager is not None: if idx < len(self.motion_blocks): module_1 = gpu_manager.managed_motion_modules[idx] diff --git a/wan/tia2mv_obj_back_id_prefix.py b/wan/tia2mv_obj_back_id_prefix.py index 7333515..7d20fa4 100644 --- a/wan/tia2mv_obj_back_id_prefix.py +++ b/wan/tia2mv_obj_back_id_prefix.py @@ -161,7 +161,8 @@ class WanTIA2MVRefBackIDPrefix: ) logging.info(f"Creating WanModel from {checkpoint_dir}") model_cfg = config - model_type = 'ti2v' if model_cfg.__name__ == 'Config: Wan TI2V 5B' else 'i2v' + #print(model_cfg.__name__) + model_type = 'ti2v' if model_cfg.__name__ == 'Config: Wan TI2V 5B' else 'i2v' #Config: Wan TI2V 5B ctx = init_empty_weights if is_accelerate_available() else nullcontext with ctx(): unet = WanModel( @@ -298,7 +299,8 @@ class WanTIA2MVRefBackIDPrefix: - H: Frame height (from max_area) - W: Frame width from max_area) """ - + up_scale = 2.75 if kwargs.get('short_side', 704)==704 else 2.0 + #print(f'frame_num: {frame_num}') if self.origin_mode: cond_image = TF.to_tensor(img).sub_(0.5).div_(0.5).to(self.device) cond_image = cond_image[None, :, None, :, :] @@ -398,9 +400,10 @@ class WanTIA2MVRefBackIDPrefix: if self.origin_mode: if mode in ['a2v','a2mv','mv','i2v']: cond_pose_sequence = torch.zeros_like(cond_pose_sequence) - + #print(f'mode: {mode}') if mode in ['p2v','mv','i2v']: audio_embs = zero_audio_embs + if self.origin_mode: h, w = cond_image.shape[-2], cond_image.shape[-1] lat_h, lat_w = h // self.vae_stride[1], w // self.vae_stride[2] @@ -411,12 +414,13 @@ class WanTIA2MVRefBackIDPrefix: lat_h=kwargs.get('lat_h', None) lat_w=kwargs.get('lat_w', None) max_seq_len=kwargs.get('max_seq_len', None) + #print(f'max_seq_len: {max_seq_len}',lat_h,lat_w) # 15232 32 56 34*32/2*56/2= 15232 noise = torch.randn( 1, 48, (frame_num - 1) // 4 + 1 + 1, lat_h, lat_w, dtype=torch.float32, - device=self.device) + device=self.device) # 初始噪声加多4帧 if self.origin_mode: cond_image = self.vae.encode([cond_image.squeeze(0).to(torch.float32)])[0].unsqueeze(0) obj_image = self.vae.encode([obj_image.repeat(1, 1, 4, 1, 1).squeeze(0).to(torch.float32)])[0].unsqueeze(0) @@ -437,7 +441,7 @@ class WanTIA2MVRefBackIDPrefix: motion_lat_h=kwargs.get('motion_lat_h', None) motion_lat_w=kwargs.get('motion_lat_w', None) motion_max_seq_len=kwargs.get('motion_max_seq_len', None) - + #print(f'motion_max_seq_len: {motion_max_seq_len}',motion_lat_h,motion_lat_w) #motion_max_seq_len: 2496 24 16 motion_noise = torch.randn( 1, 48, (frame_num - 1) // 4 + 1 + 1, motion_lat_h, @@ -479,14 +483,15 @@ class WanTIA2MVRefBackIDPrefix: else: gpu_manager = BlockGPUManager(device="cuda") gpu_manager.setup_for_inference(self.model) - #print(latent.shape,cond_image.shape,obj_image.shape) #torch.Size([1, 48, 22, 48, 32]) torch.Size([1, 48, 1, 48, 32]) torch.Size([1, 48, 1, 48, 32]) + #print(latent.shape,cond_image.shape,obj_image.shape) #torch.Size([1, 48, 35, 32, 56]) torch.Size([1, 48, 1, 32, 56]) torch.Size([1, 48, 1, 32, 56]) latent[:, :, 0:1] = cond_image latent[:, :, -1:] = obj_image - #print(latent_motion.shape,cond_dw_img.shape,small_img.shape) #torch.Size([1, 48, 22, 24, 16]) torch.Size([1, 48, 1, 24, 16]) torch.Size([1, 48, 1, 24, 16]) + #print(latent_motion.shape,cond_dw_img.shape,small_img.shape) #torch.Size([1, 48, 35, 16, 28]) torch.Size([1, 48, 1, 16, 28]) torch.Size([1, 48, 1, 16, 28]) latent_motion[:, :, 0:1] = cond_dw_img latent_motion[:, :, -1:] = small_img #print(cond_pose_sequence.shape) #torch.Size([1, 48, 21, 24, 16]) cond_pose_sequence = torch.concat([cond_pose_sequence, small_img], dim=2) + #print(f'cond_pose_sequence.shape: {cond_pose_sequence.shape}') #torch.Size([1, 48, 35, 16, 28]) cond_pose_sequence = cond_pose_sequence.to(latent_motion.dtype).to(self.device,cur_dtype) audio_embs = audio_embs.to(latent_motion.dtype).to(self.device,cur_dtype) # zero_audio_embs = zero_audio_embs.to(latent_motion.dtype).to(self.device) @@ -494,8 +499,8 @@ class WanTIA2MVRefBackIDPrefix: # prepare condition and uncondition configs arg_c = { 'context_list': [context[0].to(cur_dtype)], - 'seq_len': max_seq_len + lat_h * lat_w //4, - 'seq_len_motion': motion_max_seq_len + motion_lat_h * motion_lat_w //4, + 'seq_len': max_seq_len + lat_h * lat_w //4, # 补齐4帧长度 + 'seq_len_motion': motion_max_seq_len + motion_lat_h * motion_lat_w //4, # 15232+ 'mode': mode, 'skip_block': False, 'audio_embedding': audio_embs, @@ -531,11 +536,11 @@ class WanTIA2MVRefBackIDPrefix: arg_null_text['skip_block'] = False #print('no bad_cfg', timestep, bad_cfg) - + if mode not in ['mv','a2mv']: noise_pred_cond, noise_pred_cond_motion = self.model( x=latent_model_input,motion=cond_pose_sequence, - t=timestep,motion_t=zero_timestep, **arg_c,gpu_manager=gpu_manager, + t=timestep,motion_t=zero_timestep, **arg_c,gpu_manager=gpu_manager,up_scale=up_scale, ) torch_gc() noise_pred_drop_text, noise_pred_drop_text_motion = self.model( @@ -543,14 +548,14 @@ class WanTIA2MVRefBackIDPrefix: motion=cond_pose_sequence, t=timestep,motion_t=zero_timestep, **arg_null_text, - gpu_manager=gpu_manager, + gpu_manager=gpu_manager,up_scale=up_scale, ) torch_gc() else: # inference with CFG strategy noise_pred_cond, noise_pred_cond_motion = self.model( x=latent_model_input,motion=latent_model_input_motion, - t=timestep,motion_t=timestep, **arg_c,gpu_manager=gpu_manager, + t=timestep,motion_t=timestep, **arg_c,gpu_manager=gpu_manager,up_scale=up_scale, ) torch_gc() noise_pred_drop_text, noise_pred_drop_text_motion = self.model( @@ -559,7 +564,7 @@ class WanTIA2MVRefBackIDPrefix: t=timestep, motion_t=timestep, **arg_null_text, - gpu_manager=gpu_manager, + gpu_manager=gpu_manager,up_scale=up_scale, ) torch_gc() @@ -649,6 +654,7 @@ class WanTIA2MVRefBackIDPrefix: Generates video frames autoregressively with Latent Prefix Loop. Segment Length: 101 frames (26 temporal latents + 2 condition latents). """ + up_scale = 2.75 if kwargs.get('short_side', 704)==704 else 2.0 if self.origin_mode: # --- 1. 输入预处理 (Input Preprocessing) --- # 初始首帧 (Start Frame) - 仅用于第一段 @@ -996,14 +1002,14 @@ class WanTIA2MVRefBackIDPrefix: # Model Forward if three_cfg: noise_pred_cond, noise_pred_cond_motion = self.model( - x=latent_model_input, motion=motion_input, t=timestep, motion_t=motion_time, **arg_c,gpu_manager=gpu_manager) + x=latent_model_input, motion=motion_input, t=timestep, motion_t=motion_time, **arg_c,gpu_manager=gpu_manager,up_scale=up_scale,) torch_gc() noise_pred_drop_text, noise_pred_drop_text_motion = self.model( x=latent_model_input, motion=motion_input, t=timestep, motion_t=motion_time, - **arg_null_text,gpu_manager=gpu_manager + **arg_null_text,gpu_manager=gpu_manager,up_scale=up_scale ) torch_gc() noise_pred_pure_audio, noise_pred_pure_audio_motion = self.model( @@ -1011,7 +1017,7 @@ class WanTIA2MVRefBackIDPrefix: motion=motion_input, t=timestep, motion_t=motion_time, - **arg_pure_audio,gpu_manager=gpu_manager + **arg_pure_audio,gpu_manager=gpu_manager,up_scale=up_scale, ) torch_gc() @@ -1023,11 +1029,11 @@ class WanTIA2MVRefBackIDPrefix: ) else: noise_pred_cond, noise_pred_cond_motion = self.model( - x=latent_model_input, motion=motion_input, t=timestep, motion_t=motion_time, **arg_c,gpu_manager=gpu_manager) + x=latent_model_input, motion=motion_input, t=timestep, motion_t=motion_time, **arg_c,gpu_manager=gpu_manager,up_scale=up_scale,) torch_gc() noise_pred_drop_text, noise_pred_drop_text_motion = self.model( x=latent_model_input, motion=motion_input, t=timestep, - motion_t=motion_time, **arg_null_text,gpu_manager=gpu_manager) + motion_t=motion_time, **arg_null_text,gpu_manager=gpu_manager,up_scale=up_scale,) torch_gc() noise_pred = noise_pred_drop_text + text_guide_scale * (noise_pred_cond - noise_pred_drop_text)