diff --git a/examples/pyramidflow_img2vid.json b/examples/pyramidflow_img2vid.json new file mode 100644 index 0000000..225beee --- /dev/null +++ b/examples/pyramidflow_img2vid.json @@ -0,0 +1,576 @@ +{ + "last_node_id": 20, + "last_link_id": 27, + "nodes": [ + { + "id": 17, + "type": "LoadImage", + "pos": { + "0": 549, + "1": 1014 + }, + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 21 + ], + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "sd3stag.png", + "image" + ] + }, + { + "id": 18, + "type": "ImageResizeKJ", + "pos": { + "0": 907, + "1": 1063 + }, + "size": { + "0": 315, + "1": 266 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 21 + }, + { + "name": "get_image_size", + "type": "IMAGE", + "link": null, + "shape": 7 + }, + { + "name": "width_input", + "type": "INT", + "link": null, + "widget": { + "name": "width_input" + }, + "shape": 7 + }, + { + "name": "height_input", + "type": "INT", + "link": null, + "widget": { + "name": "height_input" + }, + "shape": 7 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 24 + ], + "slot_index": 0 + }, + { + "name": "width", + "type": "INT", + "links": null + }, + { + "name": "height", + "type": "INT", + "links": null + } + ], + "properties": { + "Node name for S&R": "ImageResizeKJ" + }, + "widgets_values": [ + 640, + 384, + "lanczos", + false, + 2, + 0, + 0, + "disabled" + ] + }, + { + "id": 19, + "type": "PyramidFlowVAEEncode", + "pos": { + "0": 1077, + "1": 918 + }, + "size": { + "0": 277.20001220703125, + "1": 46 + }, + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "PYRAMIDFLOWMODEL", + "link": 23 + }, + { + "name": "image", + "type": "IMAGE", + "link": 24 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "links": [ + 25 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "PyramidFlowVAEEncode" + }, + "widgets_values": [] + }, + { + "id": 8, + "type": "PyramidFlowVAEDecode", + "pos": { + "0": 1555, + "1": 510 + }, + "size": { + "0": 315, + "1": 102 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "PYRAMIDFLOWMODEL", + "link": 8 + }, + { + "name": "samples", + "type": "LATENT", + "link": 9 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 26 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "PyramidFlowVAEDecode" + }, + "widgets_values": [ + 256, + 2 + ] + }, + { + "id": 20, + "type": "GetImageSizeAndCount", + "pos": { + "0": 1593, + "1": 689 + }, + "size": { + "0": 277.20001220703125, + "1": 86 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 26 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 27 + ], + "slot_index": 0 + }, + { + "name": "640 width", + "type": "INT", + "links": null + }, + { + "name": "384 height", + "type": "INT", + "links": null + }, + { + "name": "65 count", + "type": "INT", + "links": null + } + ], + "properties": { + "Node name for S&R": "GetImageSizeAndCount" + }, + "widgets_values": [] + }, + { + "id": 14, + "type": "VHS_VideoCombine", + "pos": { + "0": 1916, + "1": 512 + }, + "size": [ + 750.8298728310601, + 762.497923698636 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 27 + }, + { + "name": "audio", + "type": "AUDIO", + "link": null, + "shape": 7 + }, + { + "name": "meta_batch", + "type": "VHS_BatchManager", + "link": null, + "shape": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 8, + "loop_count": 0, + "filename_prefix": "AnimateDiff", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "AnimateDiff_00009.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 8 + }, + "muted": false + } + } + }, + { + "id": 16, + "type": "PyramidFlowTextEncode", + "pos": { + "0": 551, + "1": 731 + }, + "size": { + "0": 403.6938171386719, + "1": 219.07676696777344 + }, + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "PYRAMIDFLOWMODEL", + "link": 18 + } + ], + "outputs": [ + { + "name": "prompt_embeds", + "type": "PYRAMIDFLOWPROMPT", + "links": [ + 19 + ] + } + ], + "properties": { + "Node name for S&R": "PyramidFlowTextEncode" + }, + "widgets_values": [ + "camera rotating around a stag, hyper quality, Ultra HD, 8K", + "", + false + ] + }, + { + "id": 9, + "type": "PyramidFlowSampler", + "pos": { + "0": 1077, + "1": 509 + }, + "size": { + "0": 411.5168151855469, + "1": 314 + }, + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "PYRAMIDFLOWMODEL", + "link": 7 + }, + { + "name": "prompt_embeds", + "type": "PYRAMIDFLOWPROMPT", + "link": 19 + }, + { + "name": "input_latent", + "type": "LATENT", + "link": 25, + "shape": 7 + } + ], + "outputs": [ + { + "name": "model", + "type": "PYRAMIDFLOWMODEL", + "links": [ + 8 + ] + }, + { + "name": "samples", + "type": "LATENT", + "links": [ + 9 + ], + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "PyramidFlowSampler" + }, + "widgets_values": [ + 640, + 384, + 20, + 10, + 8, + 6, + 4, + 44664248661374, + "fixed", + "" + ] + }, + { + "id": 5, + "type": "DownloadAndLoadPyramidFlowModel", + "pos": { + "0": 564, + "1": 509 + }, + "size": { + "0": 361.3862609863281, + "1": 154 + }, + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pyramidflow_model", + "type": "PYRAMIDFLOWMODEL", + "links": [ + 7, + 18, + 23 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "DownloadAndLoadPyramidFlowModel" + }, + "widgets_values": [ + "rain1011/pyramid-flow-sd3", + "diffusion_transformer_384p", + "fp32", + "fp16", + "bf16" + ] + } + ], + "links": [ + [ + 7, + 5, + 0, + 9, + 0, + "PYRAMIDFLOWMODEL" + ], + [ + 8, + 9, + 0, + 8, + 0, + "PYRAMIDFLOWMODEL" + ], + [ + 9, + 9, + 1, + 8, + 1, + "LATENT" + ], + [ + 18, + 5, + 0, + 16, + 0, + "PYRAMIDFLOWMODEL" + ], + [ + 19, + 16, + 0, + 9, + 1, + "PYRAMIDFLOWPROMPT" + ], + [ + 21, + 17, + 0, + 18, + 0, + "IMAGE" + ], + [ + 23, + 5, + 0, + 19, + 0, + "PYRAMIDFLOWMODEL" + ], + [ + 24, + 18, + 0, + 19, + 1, + "IMAGE" + ], + [ + 25, + 19, + 0, + 9, + 2, + "LATENT" + ], + [ + 26, + 8, + 0, + 20, + 0, + "IMAGE" + ], + [ + 27, + 20, + 0, + 14, + 0, + "IMAGE" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 1.0152559799479295, + "offset": [ + -470.81907375647995, + -347.52728270498267 + ] + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/nodes.py b/nodes.py index 91c88ad..f09978b 100644 --- a/nodes.py +++ b/nodes.py @@ -166,11 +166,11 @@ class PyramidFlowSampler: "prompt_embeds": ("PYRAMIDFLOWPROMPT",), "width": ("INT", {"default": 640, "min": 128, "max": 2048, "step": 8}), "height": ("INT", {"default": 384, "min": 128, "max": 2048, "step": 8}), - "steps": ("INT", {"default": 20, "min": 1, "max": 200, "step": 1}), - "video_steps": ("INT", {"default": 10, "min": 5, "max": 2048, "step": 4}), + "first_frame_steps": ("INT", {"default": 20, "min": 1, "max": 200, "step": 1, "tooltip": "Number of steps for the first frame, no effect when using input_latent"}), + "video_steps": ("INT", {"default": 10, "min": 1, "max": 2048, "step": 1, "tooltip": "Number of steps for the video latents"}), "temp": ("INT", {"default": 8, "min": 1, "tooltip": "temp=16: 5s, temp=31: 10s"}), "guidance_scale": ("FLOAT", {"default": 9.0, "min": 0.0, "max": 30.0, "step": 0.01, "tooltip": "The guidance for the first frame"}), - "video_guidance_scale": ("FLOAT", {"default": 5.0, "min": 0.0, "max": 30.0, "step": 0.01, "tooltip": "The guidance for the other video latent"}), + "video_guidance_scale": ("FLOAT", {"default": 5.0, "min": 0.0, "max": 30.0, "step": 0.01, "tooltip": "The guidance for the video latents"}), "seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}), "keep_model_loaded": ("BOOLEAN", {"default": False}), @@ -185,7 +185,7 @@ class PyramidFlowSampler: FUNCTION = "sample" CATEGORY = "PyramidFlowWrapper" - def sample(self, model, steps, prompt_embeds, seed, height, width, video_steps, temp, guidance_scale, video_guidance_scale, + def sample(self, model, first_frame_steps, prompt_embeds, seed, height, width, video_steps, temp, guidance_scale, video_guidance_scale, keep_model_loaded, input_latent=None): mm.soft_empty_cache() @@ -203,7 +203,7 @@ class PyramidFlowSampler: latents = model["model"].generate( prompt_embeds_dict = prompt_embeds, device=device, - num_inference_steps=[steps, steps, steps], #why's this a list + num_inference_steps=[first_frame_steps, first_frame_steps, first_frame_steps], #why's this a list video_num_inference_steps=[video_steps, video_steps, video_steps], #why's this a list height=height, width=width, @@ -218,7 +218,7 @@ class PyramidFlowSampler: prompt_embeds_dict = prompt_embeds, input_image_latent=input_latent, device=device, - num_inference_steps=[steps, steps, steps], #why's this a list + num_inference_steps=[video_steps, video_steps, video_steps], #why's this a list height=height, width=width, temp=temp,