diff --git a/README.md b/README.md index 93c1c20..0e774ac 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,83 @@ # ComfyUI-DiffSynth-Studio make [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio) avialbe in ComfyUI +
+
+ webpage +
+
-# developing \ No newline at end of file +## how to use +make sure `ffmpeg` is worked in your commandline +for Linux +``` +apt update +apt install ffmpeg +``` +for Windows,you can install `ffmpeg` by [WingetUI](https://github.com/marticliment/WingetUI) automatically + +then! +``` +## insatll xformers match your torch,for torch==2.1.0+cu121 +pip install xformers==0.0.22.post7 +pip install accelerate +# in ComfyUI/custom_nodes +git clone https://github.com/AIFSH/ComfyUI-DiffSynth-Studio.git +cd ComfyUI-DiffSynth-Studio +pip install -r requirements.txt +``` +weights will be downloaded from huggingface + + +## Nodes Detail and Workflow +### ExVideo Node +commig soon +### image synthesis +comming soon +### Diffutoon Node +[Diffutoon workflow](diffutoon_workflow.json) +``` +"required":{ + "source_video_path": ("VIDEO",), + "sd_model_path":("SD_MODEL_PATH",), + "postive_prompt":("TEXT",), + "negative_prompt":("TEXT",), + "start":("INT",{ + "default": 0 ## from which second of your video to be shaded + }), + "length":("INT",{ + "default": -1 ## how long you want to shade, -1 name the whole video frames + }), + "seed":("INT",{ + "default": 42 + }), + "cfg_scale":("INT",{ + "default": 3 + }), + "num_inference_steps":("INT",{ + "default": 10 + }), + "animatediff_batch_size":("INT",{ + "default": 32 ## lower it till you can run + }), + "animatediff_stride":("INT",{ + "default": 16 ## lower it till you can run + }), + "vram_limit_level":("INT",{ + "default": 0 ## meet killed? try to 1 + }), +}, +"optional":{ + "controlnet1":("ControlNetConfigUnit",), + "controlnet2":("ControlNetConfigUnit",), + "controlnet3":("ControlNetConfigUnit",), +} +``` + +### Video Stylization + +### Chinese Models + +## ask for answer as soon as you want +wechat: aifsh_98 +need donate if you mand it, +but please feel free to new issue for answering \ No newline at end of file diff --git a/__init__.py b/__init__.py index a8f9141..bd8ecbe 100644 --- a/__init__.py +++ b/__init__.py @@ -1,5 +1,5 @@ from .util_nodes import LoadVideo,PreViewVideo -from .studio_nodes import DiffTextNode,VideoShadeNode,SDPathLoader,ControlNetPathLoader +from .studio_nodes import DiffTextNode,DiffutoonNode,SDPathLoader,ControlNetPathLoader WEB_DIRECTORY = "./web" # A dictionary that contains all nodes you want to export with their names # NOTE: names should be globally unique @@ -8,7 +8,7 @@ NODE_CLASS_MAPPINGS = { "PreViewVideo": PreViewVideo, "SDPathLoader": SDPathLoader, "DiffTextNode": DiffTextNode, - "VideoShadeNode": VideoShadeNode, + "DiffutoonNode": DiffutoonNode, "ControlNetPathLoader": ControlNetPathLoader } @@ -18,6 +18,6 @@ NODE_DISPLAY_NAME_MAPPINGS = { "PreViewVideo": "PreViewVideo", "SDPathLoader": "SDPathLoader", "DiffTextNode": "DiffTextNode", - "VideoShadeNode": "VideoShadeNode", + "DiffutoonNode": "DiffutoonNode", "ControlNetPathLoader": "ControlNetPathLoader" } \ No newline at end of file diff --git a/diffsynth/controlnets/processors.py b/diffsynth/controlnets/processors.py index a378842..a88812c 100644 --- a/diffsynth/controlnets/processors.py +++ b/diffsynth/controlnets/processors.py @@ -1,3 +1,5 @@ +import os +import folder_paths from typing_extensions import Literal, TypeAlias import warnings with warnings.catch_warnings(): @@ -6,13 +8,13 @@ with warnings.catch_warnings(): CannyDetector, MidasDetector, HEDdetector, LineartDetector, LineartAnimeDetector, OpenposeDetector ) - +annotators_dir = os.path.join(folder_paths.models_dir, "Annotators") Processor_id: TypeAlias = Literal[ "canny", "depth", "softedge", "lineart", "lineart_anime", "openpose", "tile" ] class Annotator: - def __init__(self, processor_id: Processor_id, model_path="models/Annotators", detect_resolution=None): + def __init__(self, processor_id: Processor_id, model_path=annotators_dir, detect_resolution=None): if processor_id == "canny": self.processor = CannyDetector() elif processor_id == "depth": diff --git a/diffsynth/data/video.py b/diffsynth/data/video.py index 16e1918..d932425 100644 --- a/diffsynth/data/video.py +++ b/diffsynth/data/video.py @@ -114,7 +114,7 @@ class VideoData: if self.height is not None and self.width is not None: return self.height, self.width else: - height, width, _ = self.__getitem__(0).shape + height, width = self.__getitem__(0).size return height, width def __getitem__(self, item): diff --git a/diffutoon_workflow.json b/diffutoon_workflow.json new file mode 100644 index 0000000..ea10489 --- /dev/null +++ b/diffutoon_workflow.json @@ -0,0 +1,382 @@ +{ + "last_node_id": 17, + "last_link_id": 21, + "nodes": [ + { + "id": 10, + "type": "PreViewVideo", + "pos": [ + 1017.6000366210938, + 406.20001220703125 + ], + "size": { + "0": 210, + "1": 26 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "video", + "type": "VIDEO", + "link": 21 + } + ], + "properties": { + "Node name for S&R": "PreViewVideo" + } + }, + { + "id": 6, + "type": "DiffTextNode", + "pos": [ + 505, + 485 + ], + "size": { + "0": 400, + "1": 200 + }, + "flags": {}, + "order": 0, + "mode": 0, + "outputs": [ + { + "name": "TEXT", + "type": "TEXT", + "links": [ + 18 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "DiffTextNode" + }, + "widgets_values": [ + "verybadimagenegative_v1.3" + ] + }, + { + "id": 16, + "type": "ControlNetPathLoader", + "pos": [ + 1138, + 206 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "ControlNetConfigUnit", + "type": "ControlNetConfigUnit", + "links": [ + 20 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "ControlNetPathLoader" + }, + "widgets_values": [ + "tile", + 0.5, + null + ] + }, + { + "id": 15, + "type": "ControlNetPathLoader", + "pos": [ + 1128, + 17 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "ControlNetConfigUnit", + "type": "ControlNetConfigUnit", + "links": [ + 19 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "ControlNetPathLoader" + }, + "widgets_values": [ + "lineart", + 0.5, + "control_v11p_sd15_lineart.pth" + ] + }, + { + "id": 3, + "type": "LoadVideo", + "pos": [ + 153, + -100 + ], + "size": { + "0": 315, + "1": 383 + }, + "flags": {}, + "order": 3, + "mode": 0, + "outputs": [ + { + "name": "VIDEO", + "type": "VIDEO", + "links": [ + 15 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "LoadVideo" + }, + "widgets_values": [ + "diffutoondemo.mp4", + "Video", + { + "hidden": false, + "paused": false, + "params": {} + } + ] + }, + { + "id": 14, + "type": "SDPathLoader", + "pos": [ + 105, + 423 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 4, + "mode": 0, + "outputs": [ + { + "name": "SD_MODEL_PATH", + "type": "SD_MODEL_PATH", + "links": [ + 16 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "SDPathLoader" + }, + "widgets_values": [ + "philz1337x/flat2DAnimerge_v45Sharp", + "flat2DAnimerge_v45Sharp.safetensors", + "flat2DAnimerge_v45Sharp.safetensors" + ] + }, + { + "id": 5, + "type": "DiffTextNode", + "pos": [ + 104, + 664 + ], + "size": { + "0": 400, + "1": 200 + }, + "flags": {}, + "order": 5, + "mode": 0, + "outputs": [ + { + "name": "TEXT", + "type": "TEXT", + "links": [ + 17 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "DiffTextNode" + }, + "widgets_values": [ + "best quality, perfect anime illustration, light, a girl is dancing, smile, solo" + ] + }, + { + "id": 17, + "type": "DiffutoonNode", + "pos": [ + 674.969524572754, + 27.48489999999994 + ], + "size": { + "0": 315, + "1": 370 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "source_video_path", + "type": "VIDEO", + "link": 15 + }, + { + "name": "sd_model_path", + "type": "SD_MODEL_PATH", + "link": 16 + }, + { + "name": "postive_prompt", + "type": "TEXT", + "link": 17 + }, + { + "name": "negative_prompt", + "type": "TEXT", + "link": 18, + "slot_index": 3 + }, + { + "name": "controlnet1", + "type": "ControlNetConfigUnit", + "link": 19, + "slot_index": 4 + }, + { + "name": "controlnet2", + "type": "ControlNetConfigUnit", + "link": 20, + "slot_index": 5 + }, + { + "name": "controlnet3", + "type": "ControlNetConfigUnit", + "link": null + } + ], + "outputs": [ + { + "name": "VIDEO", + "type": "VIDEO", + "links": [ + 21 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "DiffutoonNode" + }, + "widgets_values": [ + 40, + 1, + 924, + "randomize", + 3, + 10, + 32, + 16, + 0 + ] + } + ], + "links": [ + [ + 15, + 3, + 0, + 17, + 0, + "VIDEO" + ], + [ + 16, + 14, + 0, + 17, + 1, + "SD_MODEL_PATH" + ], + [ + 17, + 5, + 0, + 17, + 2, + "TEXT" + ], + [ + 18, + 6, + 0, + 17, + 3, + "TEXT" + ], + [ + 19, + 15, + 0, + 17, + 4, + "ControlNetConfigUnit" + ], + [ + 20, + 16, + 0, + 17, + 5, + "ControlNetConfigUnit" + ], + [ + 21, + 17, + 0, + 10, + 0, + "VIDEO" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.6830134553650705, + "offset": [ + 88.7050890441895, + 88.17900000000009 + ] + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/studio_nodes.py b/studio_nodes.py index 860520b..037b261 100644 --- a/studio_nodes.py +++ b/studio_nodes.py @@ -1,6 +1,6 @@ import os,sys import shutil -import time +import time,math import torch from .util_nodes import now_dir,output_dir sys.path.append(os.path.join(now_dir)) @@ -14,17 +14,14 @@ from huggingface_hub import hf_hub_download models_dir = os.path.join(now_dir, "models") animatediff_dir = os.path.join(models_dir,"AnimateDiff") -annotators_dir = os.path.join(models_dir, "Annotators") +annotators_dir = os.path.join(folder_paths.models_dir, "Annotators") textual_inversion_dir = os.path.join(models_dir, "textual_inversion") rife_dir = os.path.join(models_dir, "RIFE") device = "cuda" if cuda_malloc.cuda_malloc_supported() else "cpu" -def get_4x_num(num): - num_ = round(num) - while num_ % 4 != 0: - num_ -= 1 - return num_ +def get_64x_num(num): + return math.ceil(num / 64) * 64 class DiffTextNode: @classmethod @@ -126,13 +123,14 @@ class ControlNetPathLoader: } return (out_dict,) -class VideoShadeNode: +class DiffutoonNode: def __init__(self): try: # AnimateDiff hf_hub_download(repo_id="guoyww/animatediff",filename="mm_sd_v15_v2.ckpt",local_dir=animatediff_dir) - # ControlNet - + # Annotators + hf_hub_download(repo_id="lllyasviel/Annotators",filename="sk_model.pth",local_dir=annotators_dir) + hf_hub_download(repo_id="lllyasviel/Annotators",filename="sk_model2.pth",local_dir=annotators_dir) #textual_inversion hf_hub_download(repo_id="gemasai/verybadimagenegative_v1.3",filename="verybadimagenegative_v1.3.pt",local_dir=textual_inversion_dir) # RIFE @@ -164,10 +162,13 @@ class VideoShadeNode: "default": 10 }), "animatediff_batch_size":("INT",{ - "default": 32 + "default": 4 }), "animatediff_stride":("INT",{ - "default": 16 + "default": 2 + }), + "vram_limit_level":("INT",{ + "default": 0 }), }, "optional":{ @@ -188,7 +189,7 @@ class VideoShadeNode: def maketoon(self,source_video_path,sd_model_path,postive_prompt,negative_prompt,start,length,seed, cfg_scale,num_inference_steps,animatediff_batch_size,animatediff_stride, - controlnet1=None,controlnet2=None,controlnet3=None,): + vram_limit_level,controlnet1=None,controlnet2=None,controlnet3=None,): # load models model_manager = ModelManager(torch_dtype=torch.float16, device=device) shutil.rmtree(os.path.join(textual_inversion_dir,".huggingface"),ignore_errors=True) @@ -219,19 +220,21 @@ class VideoShadeNode: # The original video is here: https://www.bilibili.com/video/BV19w411A7YJ/ video = VideoData(video_file=source_video_path) - org_h,org_w = video.shape - height, width = (1024,get_4x_num(1024*org_w/org_h)) if org_h > org_w else (get_4x_num(1024*org_h/org_w),1024) - print(f"orginal size: {org_h}X{org_w} \t resize: {height}X{width}") + org_w, org_h = video.shape() + height, width = (1024,get_64x_num(1024*org_w/org_h)) if org_h > org_w else (get_64x_num(1024*org_h/org_w),1024) + print(f"orginal size: {org_w}X{org_h} resize to: {height}X{width}") video.set_shape(height,width) - fps = video.data.reader.get_meta_data['fps'] - duration = video.data.reader.get_meta_data['duration'] + video_meta_data = video.data.reader.get_meta_data() + fps = round(video_meta_data['fps']) + duration = round(video_meta_data['duration']) + print(f"orginal fps: {fps} duration: {duration}") assert start < duration and start + length < duration if length == -1: - input_video = [video[i] for i in range(start*fps, (duration-start)*fps)] + input_video = [video[i] for i in range(start*fps, len(video))] else: input_video = [video[i] for i in range(start*fps, (start+length)*fps)] - + print(f"{len(input_video)} frame will be to shade") # Toon shading (20G VRAM) torch.manual_seed(seed) output_video = pipe( @@ -241,7 +244,7 @@ class VideoShadeNode: controlnet_frames=input_video, num_frames=len(input_video), num_inference_steps=num_inference_steps, height=height, width=width, animatediff_batch_size=animatediff_batch_size, animatediff_stride=animatediff_stride, - vram_limit_level=0, + vram_limit_level=vram_limit_level, ) output_video = smoother(output_video)