diff --git a/README.md b/README.md
index 93c1c20..0e774ac 100644
--- a/README.md
+++ b/README.md
@@ -1,4 +1,83 @@
# ComfyUI-DiffSynth-Studio
make [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio) avialbe in ComfyUI
+
+
+
+
+
-# developing
\ No newline at end of file
+## how to use
+make sure `ffmpeg` is worked in your commandline
+for Linux
+```
+apt update
+apt install ffmpeg
+```
+for Windows,you can install `ffmpeg` by [WingetUI](https://github.com/marticliment/WingetUI) automatically
+
+then!
+```
+## insatll xformers match your torch,for torch==2.1.0+cu121
+pip install xformers==0.0.22.post7
+pip install accelerate
+# in ComfyUI/custom_nodes
+git clone https://github.com/AIFSH/ComfyUI-DiffSynth-Studio.git
+cd ComfyUI-DiffSynth-Studio
+pip install -r requirements.txt
+```
+weights will be downloaded from huggingface
+
+
+## Nodes Detail and Workflow
+### ExVideo Node
+commig soon
+### image synthesis
+comming soon
+### Diffutoon Node
+[Diffutoon workflow](diffutoon_workflow.json)
+```
+"required":{
+ "source_video_path": ("VIDEO",),
+ "sd_model_path":("SD_MODEL_PATH",),
+ "postive_prompt":("TEXT",),
+ "negative_prompt":("TEXT",),
+ "start":("INT",{
+ "default": 0 ## from which second of your video to be shaded
+ }),
+ "length":("INT",{
+ "default": -1 ## how long you want to shade, -1 name the whole video frames
+ }),
+ "seed":("INT",{
+ "default": 42
+ }),
+ "cfg_scale":("INT",{
+ "default": 3
+ }),
+ "num_inference_steps":("INT",{
+ "default": 10
+ }),
+ "animatediff_batch_size":("INT",{
+ "default": 32 ## lower it till you can run
+ }),
+ "animatediff_stride":("INT",{
+ "default": 16 ## lower it till you can run
+ }),
+ "vram_limit_level":("INT",{
+ "default": 0 ## meet killed? try to 1
+ }),
+},
+"optional":{
+ "controlnet1":("ControlNetConfigUnit",),
+ "controlnet2":("ControlNetConfigUnit",),
+ "controlnet3":("ControlNetConfigUnit",),
+}
+```
+
+### Video Stylization
+
+### Chinese Models
+
+## ask for answer as soon as you want
+wechat: aifsh_98
+need donate if you mand it,
+but please feel free to new issue for answering
\ No newline at end of file
diff --git a/__init__.py b/__init__.py
index a8f9141..bd8ecbe 100644
--- a/__init__.py
+++ b/__init__.py
@@ -1,5 +1,5 @@
from .util_nodes import LoadVideo,PreViewVideo
-from .studio_nodes import DiffTextNode,VideoShadeNode,SDPathLoader,ControlNetPathLoader
+from .studio_nodes import DiffTextNode,DiffutoonNode,SDPathLoader,ControlNetPathLoader
WEB_DIRECTORY = "./web"
# A dictionary that contains all nodes you want to export with their names
# NOTE: names should be globally unique
@@ -8,7 +8,7 @@ NODE_CLASS_MAPPINGS = {
"PreViewVideo": PreViewVideo,
"SDPathLoader": SDPathLoader,
"DiffTextNode": DiffTextNode,
- "VideoShadeNode": VideoShadeNode,
+ "DiffutoonNode": DiffutoonNode,
"ControlNetPathLoader": ControlNetPathLoader
}
@@ -18,6 +18,6 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"PreViewVideo": "PreViewVideo",
"SDPathLoader": "SDPathLoader",
"DiffTextNode": "DiffTextNode",
- "VideoShadeNode": "VideoShadeNode",
+ "DiffutoonNode": "DiffutoonNode",
"ControlNetPathLoader": "ControlNetPathLoader"
}
\ No newline at end of file
diff --git a/diffsynth/controlnets/processors.py b/diffsynth/controlnets/processors.py
index a378842..a88812c 100644
--- a/diffsynth/controlnets/processors.py
+++ b/diffsynth/controlnets/processors.py
@@ -1,3 +1,5 @@
+import os
+import folder_paths
from typing_extensions import Literal, TypeAlias
import warnings
with warnings.catch_warnings():
@@ -6,13 +8,13 @@ with warnings.catch_warnings():
CannyDetector, MidasDetector, HEDdetector, LineartDetector, LineartAnimeDetector, OpenposeDetector
)
-
+annotators_dir = os.path.join(folder_paths.models_dir, "Annotators")
Processor_id: TypeAlias = Literal[
"canny", "depth", "softedge", "lineart", "lineart_anime", "openpose", "tile"
]
class Annotator:
- def __init__(self, processor_id: Processor_id, model_path="models/Annotators", detect_resolution=None):
+ def __init__(self, processor_id: Processor_id, model_path=annotators_dir, detect_resolution=None):
if processor_id == "canny":
self.processor = CannyDetector()
elif processor_id == "depth":
diff --git a/diffsynth/data/video.py b/diffsynth/data/video.py
index 16e1918..d932425 100644
--- a/diffsynth/data/video.py
+++ b/diffsynth/data/video.py
@@ -114,7 +114,7 @@ class VideoData:
if self.height is not None and self.width is not None:
return self.height, self.width
else:
- height, width, _ = self.__getitem__(0).shape
+ height, width = self.__getitem__(0).size
return height, width
def __getitem__(self, item):
diff --git a/diffutoon_workflow.json b/diffutoon_workflow.json
new file mode 100644
index 0000000..ea10489
--- /dev/null
+++ b/diffutoon_workflow.json
@@ -0,0 +1,382 @@
+{
+ "last_node_id": 17,
+ "last_link_id": 21,
+ "nodes": [
+ {
+ "id": 10,
+ "type": "PreViewVideo",
+ "pos": [
+ 1017.6000366210938,
+ 406.20001220703125
+ ],
+ "size": {
+ "0": 210,
+ "1": 26
+ },
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "video",
+ "type": "VIDEO",
+ "link": 21
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "PreViewVideo"
+ }
+ },
+ {
+ "id": 6,
+ "type": "DiffTextNode",
+ "pos": [
+ 505,
+ 485
+ ],
+ "size": {
+ "0": 400,
+ "1": 200
+ },
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "TEXT",
+ "type": "TEXT",
+ "links": [
+ 18
+ ],
+ "shape": 3
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "DiffTextNode"
+ },
+ "widgets_values": [
+ "verybadimagenegative_v1.3"
+ ]
+ },
+ {
+ "id": 16,
+ "type": "ControlNetPathLoader",
+ "pos": [
+ 1138,
+ 206
+ ],
+ "size": {
+ "0": 315,
+ "1": 106
+ },
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "ControlNetConfigUnit",
+ "type": "ControlNetConfigUnit",
+ "links": [
+ 20
+ ],
+ "shape": 3
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ControlNetPathLoader"
+ },
+ "widgets_values": [
+ "tile",
+ 0.5,
+ null
+ ]
+ },
+ {
+ "id": 15,
+ "type": "ControlNetPathLoader",
+ "pos": [
+ 1128,
+ 17
+ ],
+ "size": {
+ "0": 315,
+ "1": 106
+ },
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "ControlNetConfigUnit",
+ "type": "ControlNetConfigUnit",
+ "links": [
+ 19
+ ],
+ "shape": 3
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ControlNetPathLoader"
+ },
+ "widgets_values": [
+ "lineart",
+ 0.5,
+ "control_v11p_sd15_lineart.pth"
+ ]
+ },
+ {
+ "id": 3,
+ "type": "LoadVideo",
+ "pos": [
+ 153,
+ -100
+ ],
+ "size": {
+ "0": 315,
+ "1": 383
+ },
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "VIDEO",
+ "type": "VIDEO",
+ "links": [
+ 15
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "LoadVideo"
+ },
+ "widgets_values": [
+ "diffutoondemo.mp4",
+ "Video",
+ {
+ "hidden": false,
+ "paused": false,
+ "params": {}
+ }
+ ]
+ },
+ {
+ "id": 14,
+ "type": "SDPathLoader",
+ "pos": [
+ 105,
+ 423
+ ],
+ "size": {
+ "0": 315,
+ "1": 106
+ },
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "SD_MODEL_PATH",
+ "type": "SD_MODEL_PATH",
+ "links": [
+ 16
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "SDPathLoader"
+ },
+ "widgets_values": [
+ "philz1337x/flat2DAnimerge_v45Sharp",
+ "flat2DAnimerge_v45Sharp.safetensors",
+ "flat2DAnimerge_v45Sharp.safetensors"
+ ]
+ },
+ {
+ "id": 5,
+ "type": "DiffTextNode",
+ "pos": [
+ 104,
+ 664
+ ],
+ "size": {
+ "0": 400,
+ "1": 200
+ },
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "TEXT",
+ "type": "TEXT",
+ "links": [
+ 17
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "DiffTextNode"
+ },
+ "widgets_values": [
+ "best quality, perfect anime illustration, light, a girl is dancing, smile, solo"
+ ]
+ },
+ {
+ "id": 17,
+ "type": "DiffutoonNode",
+ "pos": [
+ 674.969524572754,
+ 27.48489999999994
+ ],
+ "size": {
+ "0": 315,
+ "1": 370
+ },
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "source_video_path",
+ "type": "VIDEO",
+ "link": 15
+ },
+ {
+ "name": "sd_model_path",
+ "type": "SD_MODEL_PATH",
+ "link": 16
+ },
+ {
+ "name": "postive_prompt",
+ "type": "TEXT",
+ "link": 17
+ },
+ {
+ "name": "negative_prompt",
+ "type": "TEXT",
+ "link": 18,
+ "slot_index": 3
+ },
+ {
+ "name": "controlnet1",
+ "type": "ControlNetConfigUnit",
+ "link": 19,
+ "slot_index": 4
+ },
+ {
+ "name": "controlnet2",
+ "type": "ControlNetConfigUnit",
+ "link": 20,
+ "slot_index": 5
+ },
+ {
+ "name": "controlnet3",
+ "type": "ControlNetConfigUnit",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "VIDEO",
+ "type": "VIDEO",
+ "links": [
+ 21
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "DiffutoonNode"
+ },
+ "widgets_values": [
+ 40,
+ 1,
+ 924,
+ "randomize",
+ 3,
+ 10,
+ 32,
+ 16,
+ 0
+ ]
+ }
+ ],
+ "links": [
+ [
+ 15,
+ 3,
+ 0,
+ 17,
+ 0,
+ "VIDEO"
+ ],
+ [
+ 16,
+ 14,
+ 0,
+ 17,
+ 1,
+ "SD_MODEL_PATH"
+ ],
+ [
+ 17,
+ 5,
+ 0,
+ 17,
+ 2,
+ "TEXT"
+ ],
+ [
+ 18,
+ 6,
+ 0,
+ 17,
+ 3,
+ "TEXT"
+ ],
+ [
+ 19,
+ 15,
+ 0,
+ 17,
+ 4,
+ "ControlNetConfigUnit"
+ ],
+ [
+ 20,
+ 16,
+ 0,
+ 17,
+ 5,
+ "ControlNetConfigUnit"
+ ],
+ [
+ 21,
+ 17,
+ 0,
+ 10,
+ 0,
+ "VIDEO"
+ ]
+ ],
+ "groups": [],
+ "config": {},
+ "extra": {
+ "ds": {
+ "scale": 0.6830134553650705,
+ "offset": [
+ 88.7050890441895,
+ 88.17900000000009
+ ]
+ }
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/studio_nodes.py b/studio_nodes.py
index 860520b..037b261 100644
--- a/studio_nodes.py
+++ b/studio_nodes.py
@@ -1,6 +1,6 @@
import os,sys
import shutil
-import time
+import time,math
import torch
from .util_nodes import now_dir,output_dir
sys.path.append(os.path.join(now_dir))
@@ -14,17 +14,14 @@ from huggingface_hub import hf_hub_download
models_dir = os.path.join(now_dir, "models")
animatediff_dir = os.path.join(models_dir,"AnimateDiff")
-annotators_dir = os.path.join(models_dir, "Annotators")
+annotators_dir = os.path.join(folder_paths.models_dir, "Annotators")
textual_inversion_dir = os.path.join(models_dir, "textual_inversion")
rife_dir = os.path.join(models_dir, "RIFE")
device = "cuda" if cuda_malloc.cuda_malloc_supported() else "cpu"
-def get_4x_num(num):
- num_ = round(num)
- while num_ % 4 != 0:
- num_ -= 1
- return num_
+def get_64x_num(num):
+ return math.ceil(num / 64) * 64
class DiffTextNode:
@classmethod
@@ -126,13 +123,14 @@ class ControlNetPathLoader:
}
return (out_dict,)
-class VideoShadeNode:
+class DiffutoonNode:
def __init__(self):
try:
# AnimateDiff
hf_hub_download(repo_id="guoyww/animatediff",filename="mm_sd_v15_v2.ckpt",local_dir=animatediff_dir)
- # ControlNet
-
+ # Annotators
+ hf_hub_download(repo_id="lllyasviel/Annotators",filename="sk_model.pth",local_dir=annotators_dir)
+ hf_hub_download(repo_id="lllyasviel/Annotators",filename="sk_model2.pth",local_dir=annotators_dir)
#textual_inversion
hf_hub_download(repo_id="gemasai/verybadimagenegative_v1.3",filename="verybadimagenegative_v1.3.pt",local_dir=textual_inversion_dir)
# RIFE
@@ -164,10 +162,13 @@ class VideoShadeNode:
"default": 10
}),
"animatediff_batch_size":("INT",{
- "default": 32
+ "default": 4
}),
"animatediff_stride":("INT",{
- "default": 16
+ "default": 2
+ }),
+ "vram_limit_level":("INT",{
+ "default": 0
}),
},
"optional":{
@@ -188,7 +189,7 @@ class VideoShadeNode:
def maketoon(self,source_video_path,sd_model_path,postive_prompt,negative_prompt,start,length,seed,
cfg_scale,num_inference_steps,animatediff_batch_size,animatediff_stride,
- controlnet1=None,controlnet2=None,controlnet3=None,):
+ vram_limit_level,controlnet1=None,controlnet2=None,controlnet3=None,):
# load models
model_manager = ModelManager(torch_dtype=torch.float16, device=device)
shutil.rmtree(os.path.join(textual_inversion_dir,".huggingface"),ignore_errors=True)
@@ -219,19 +220,21 @@ class VideoShadeNode:
# The original video is here: https://www.bilibili.com/video/BV19w411A7YJ/
video = VideoData(video_file=source_video_path)
- org_h,org_w = video.shape
- height, width = (1024,get_4x_num(1024*org_w/org_h)) if org_h > org_w else (get_4x_num(1024*org_h/org_w),1024)
- print(f"orginal size: {org_h}X{org_w} \t resize: {height}X{width}")
+ org_w, org_h = video.shape()
+ height, width = (1024,get_64x_num(1024*org_w/org_h)) if org_h > org_w else (get_64x_num(1024*org_h/org_w),1024)
+ print(f"orginal size: {org_w}X{org_h} resize to: {height}X{width}")
video.set_shape(height,width)
- fps = video.data.reader.get_meta_data['fps']
- duration = video.data.reader.get_meta_data['duration']
+ video_meta_data = video.data.reader.get_meta_data()
+ fps = round(video_meta_data['fps'])
+ duration = round(video_meta_data['duration'])
+ print(f"orginal fps: {fps} duration: {duration}")
assert start < duration and start + length < duration
if length == -1:
- input_video = [video[i] for i in range(start*fps, (duration-start)*fps)]
+ input_video = [video[i] for i in range(start*fps, len(video))]
else:
input_video = [video[i] for i in range(start*fps, (start+length)*fps)]
-
+ print(f"{len(input_video)} frame will be to shade")
# Toon shading (20G VRAM)
torch.manual_seed(seed)
output_video = pipe(
@@ -241,7 +244,7 @@ class VideoShadeNode:
controlnet_frames=input_video, num_frames=len(input_video),
num_inference_steps=num_inference_steps, height=height, width=width,
animatediff_batch_size=animatediff_batch_size, animatediff_stride=animatediff_stride,
- vram_limit_level=0,
+ vram_limit_level=vram_limit_level,
)
output_video = smoother(output_video)