publish diffutoon node

This commit is contained in:
AIFSH
2024-06-27 06:13:37 +08:00
parent bce34be99e
commit 7d5f08adda
6 changed files with 494 additions and 28 deletions
+80 -1
View File
@@ -1,4 +1,83 @@
# ComfyUI-DiffSynth-Studio
make [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio) avialbe in ComfyUI
<div>
<figure>
<img alt='webpage' src="web.png?raw=true" width="600px"/>
<figure>
</div>
# developing
## how to use
make sure `ffmpeg` is worked in your commandline
for Linux
```
apt update
apt install ffmpeg
```
for Windows,you can install `ffmpeg` by [WingetUI](https://github.com/marticliment/WingetUI) automatically
then!
```
## insatll xformers match your torch,for torch==2.1.0+cu121
pip install xformers==0.0.22.post7
pip install accelerate
# in ComfyUI/custom_nodes
git clone https://github.com/AIFSH/ComfyUI-DiffSynth-Studio.git
cd ComfyUI-DiffSynth-Studio
pip install -r requirements.txt
```
weights will be downloaded from huggingface
## Nodes Detail and Workflow
### ExVideo Node
commig soon
### image synthesis
comming soon
### Diffutoon Node
[Diffutoon workflow](diffutoon_workflow.json)
```
"required":{
"source_video_path": ("VIDEO",),
"sd_model_path":("SD_MODEL_PATH",),
"postive_prompt":("TEXT",),
"negative_prompt":("TEXT",),
"start":("INT",{
"default": 0 ## from which second of your video to be shaded
}),
"length":("INT",{
"default": -1 ## how long you want to shade, -1 name the whole video frames
}),
"seed":("INT",{
"default": 42
}),
"cfg_scale":("INT",{
"default": 3
}),
"num_inference_steps":("INT",{
"default": 10
}),
"animatediff_batch_size":("INT",{
"default": 32 ## lower it till you can run
}),
"animatediff_stride":("INT",{
"default": 16 ## lower it till you can run
}),
"vram_limit_level":("INT",{
"default": 0 ## meet killed? try to 1
}),
},
"optional":{
"controlnet1":("ControlNetConfigUnit",),
"controlnet2":("ControlNetConfigUnit",),
"controlnet3":("ControlNetConfigUnit",),
}
```
### Video Stylization
### Chinese Models
## ask for answer as soon as you want
wechat: aifsh_98
need donate if you mand it,
but please feel free to new issue for answering
+3 -3
View File
@@ -1,5 +1,5 @@
from .util_nodes import LoadVideo,PreViewVideo
from .studio_nodes import DiffTextNode,VideoShadeNode,SDPathLoader,ControlNetPathLoader
from .studio_nodes import DiffTextNode,DiffutoonNode,SDPathLoader,ControlNetPathLoader
WEB_DIRECTORY = "./web"
# A dictionary that contains all nodes you want to export with their names
# NOTE: names should be globally unique
@@ -8,7 +8,7 @@ NODE_CLASS_MAPPINGS = {
"PreViewVideo": PreViewVideo,
"SDPathLoader": SDPathLoader,
"DiffTextNode": DiffTextNode,
"VideoShadeNode": VideoShadeNode,
"DiffutoonNode": DiffutoonNode,
"ControlNetPathLoader": ControlNetPathLoader
}
@@ -18,6 +18,6 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"PreViewVideo": "PreViewVideo",
"SDPathLoader": "SDPathLoader",
"DiffTextNode": "DiffTextNode",
"VideoShadeNode": "VideoShadeNode",
"DiffutoonNode": "DiffutoonNode",
"ControlNetPathLoader": "ControlNetPathLoader"
}
+4 -2
View File
@@ -1,3 +1,5 @@
import os
import folder_paths
from typing_extensions import Literal, TypeAlias
import warnings
with warnings.catch_warnings():
@@ -6,13 +8,13 @@ with warnings.catch_warnings():
CannyDetector, MidasDetector, HEDdetector, LineartDetector, LineartAnimeDetector, OpenposeDetector
)
annotators_dir = os.path.join(folder_paths.models_dir, "Annotators")
Processor_id: TypeAlias = Literal[
"canny", "depth", "softedge", "lineart", "lineart_anime", "openpose", "tile"
]
class Annotator:
def __init__(self, processor_id: Processor_id, model_path="models/Annotators", detect_resolution=None):
def __init__(self, processor_id: Processor_id, model_path=annotators_dir, detect_resolution=None):
if processor_id == "canny":
self.processor = CannyDetector()
elif processor_id == "depth":
+1 -1
View File
@@ -114,7 +114,7 @@ class VideoData:
if self.height is not None and self.width is not None:
return self.height, self.width
else:
height, width, _ = self.__getitem__(0).shape
height, width = self.__getitem__(0).size
return height, width
def __getitem__(self, item):
+382
View File
@@ -0,0 +1,382 @@
{
"last_node_id": 17,
"last_link_id": 21,
"nodes": [
{
"id": 10,
"type": "PreViewVideo",
"pos": [
1017.6000366210938,
406.20001220703125
],
"size": {
"0": 210,
"1": 26
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "video",
"type": "VIDEO",
"link": 21
}
],
"properties": {
"Node name for S&R": "PreViewVideo"
}
},
{
"id": 6,
"type": "DiffTextNode",
"pos": [
505,
485
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
18
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "DiffTextNode"
},
"widgets_values": [
"verybadimagenegative_v1.3"
]
},
{
"id": 16,
"type": "ControlNetPathLoader",
"pos": [
1138,
206
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "ControlNetConfigUnit",
"type": "ControlNetConfigUnit",
"links": [
20
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "ControlNetPathLoader"
},
"widgets_values": [
"tile",
0.5,
null
]
},
{
"id": 15,
"type": "ControlNetPathLoader",
"pos": [
1128,
17
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "ControlNetConfigUnit",
"type": "ControlNetConfigUnit",
"links": [
19
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "ControlNetPathLoader"
},
"widgets_values": [
"lineart",
0.5,
"control_v11p_sd15_lineart.pth"
]
},
{
"id": 3,
"type": "LoadVideo",
"pos": [
153,
-100
],
"size": {
"0": 315,
"1": 383
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "VIDEO",
"type": "VIDEO",
"links": [
15
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadVideo"
},
"widgets_values": [
"diffutoondemo.mp4",
"Video",
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 14,
"type": "SDPathLoader",
"pos": [
105,
423
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "SD_MODEL_PATH",
"type": "SD_MODEL_PATH",
"links": [
16
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "SDPathLoader"
},
"widgets_values": [
"philz1337x/flat2DAnimerge_v45Sharp",
"flat2DAnimerge_v45Sharp.safetensors",
"flat2DAnimerge_v45Sharp.safetensors"
]
},
{
"id": 5,
"type": "DiffTextNode",
"pos": [
104,
664
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
17
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DiffTextNode"
},
"widgets_values": [
"best quality, perfect anime illustration, light, a girl is dancing, smile, solo"
]
},
{
"id": 17,
"type": "DiffutoonNode",
"pos": [
674.969524572754,
27.48489999999994
],
"size": {
"0": 315,
"1": 370
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "source_video_path",
"type": "VIDEO",
"link": 15
},
{
"name": "sd_model_path",
"type": "SD_MODEL_PATH",
"link": 16
},
{
"name": "postive_prompt",
"type": "TEXT",
"link": 17
},
{
"name": "negative_prompt",
"type": "TEXT",
"link": 18,
"slot_index": 3
},
{
"name": "controlnet1",
"type": "ControlNetConfigUnit",
"link": 19,
"slot_index": 4
},
{
"name": "controlnet2",
"type": "ControlNetConfigUnit",
"link": 20,
"slot_index": 5
},
{
"name": "controlnet3",
"type": "ControlNetConfigUnit",
"link": null
}
],
"outputs": [
{
"name": "VIDEO",
"type": "VIDEO",
"links": [
21
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DiffutoonNode"
},
"widgets_values": [
40,
1,
924,
"randomize",
3,
10,
32,
16,
0
]
}
],
"links": [
[
15,
3,
0,
17,
0,
"VIDEO"
],
[
16,
14,
0,
17,
1,
"SD_MODEL_PATH"
],
[
17,
5,
0,
17,
2,
"TEXT"
],
[
18,
6,
0,
17,
3,
"TEXT"
],
[
19,
15,
0,
17,
4,
"ControlNetConfigUnit"
],
[
20,
16,
0,
17,
5,
"ControlNetConfigUnit"
],
[
21,
17,
0,
10,
0,
"VIDEO"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.6830134553650705,
"offset": [
88.7050890441895,
88.17900000000009
]
}
},
"version": 0.4
}
+24 -21
View File
@@ -1,6 +1,6 @@
import os,sys
import shutil
import time
import time,math
import torch
from .util_nodes import now_dir,output_dir
sys.path.append(os.path.join(now_dir))
@@ -14,17 +14,14 @@ from huggingface_hub import hf_hub_download
models_dir = os.path.join(now_dir, "models")
animatediff_dir = os.path.join(models_dir,"AnimateDiff")
annotators_dir = os.path.join(models_dir, "Annotators")
annotators_dir = os.path.join(folder_paths.models_dir, "Annotators")
textual_inversion_dir = os.path.join(models_dir, "textual_inversion")
rife_dir = os.path.join(models_dir, "RIFE")
device = "cuda" if cuda_malloc.cuda_malloc_supported() else "cpu"
def get_4x_num(num):
num_ = round(num)
while num_ % 4 != 0:
num_ -= 1
return num_
def get_64x_num(num):
return math.ceil(num / 64) * 64
class DiffTextNode:
@classmethod
@@ -126,13 +123,14 @@ class ControlNetPathLoader:
}
return (out_dict,)
class VideoShadeNode:
class DiffutoonNode:
def __init__(self):
try:
# AnimateDiff
hf_hub_download(repo_id="guoyww/animatediff",filename="mm_sd_v15_v2.ckpt",local_dir=animatediff_dir)
# ControlNet
# Annotators
hf_hub_download(repo_id="lllyasviel/Annotators",filename="sk_model.pth",local_dir=annotators_dir)
hf_hub_download(repo_id="lllyasviel/Annotators",filename="sk_model2.pth",local_dir=annotators_dir)
#textual_inversion
hf_hub_download(repo_id="gemasai/verybadimagenegative_v1.3",filename="verybadimagenegative_v1.3.pt",local_dir=textual_inversion_dir)
# RIFE
@@ -164,10 +162,13 @@ class VideoShadeNode:
"default": 10
}),
"animatediff_batch_size":("INT",{
"default": 32
"default": 4
}),
"animatediff_stride":("INT",{
"default": 16
"default": 2
}),
"vram_limit_level":("INT",{
"default": 0
}),
},
"optional":{
@@ -188,7 +189,7 @@ class VideoShadeNode:
def maketoon(self,source_video_path,sd_model_path,postive_prompt,negative_prompt,start,length,seed,
cfg_scale,num_inference_steps,animatediff_batch_size,animatediff_stride,
controlnet1=None,controlnet2=None,controlnet3=None,):
vram_limit_level,controlnet1=None,controlnet2=None,controlnet3=None,):
# load models
model_manager = ModelManager(torch_dtype=torch.float16, device=device)
shutil.rmtree(os.path.join(textual_inversion_dir,".huggingface"),ignore_errors=True)
@@ -219,19 +220,21 @@ class VideoShadeNode:
# The original video is here: https://www.bilibili.com/video/BV19w411A7YJ/
video = VideoData(video_file=source_video_path)
org_h,org_w = video.shape
height, width = (1024,get_4x_num(1024*org_w/org_h)) if org_h > org_w else (get_4x_num(1024*org_h/org_w),1024)
print(f"orginal size: {org_h}X{org_w} \t resize: {height}X{width}")
org_w, org_h = video.shape()
height, width = (1024,get_64x_num(1024*org_w/org_h)) if org_h > org_w else (get_64x_num(1024*org_h/org_w),1024)
print(f"orginal size: {org_w}X{org_h} resize to: {height}X{width}")
video.set_shape(height,width)
fps = video.data.reader.get_meta_data['fps']
duration = video.data.reader.get_meta_data['duration']
video_meta_data = video.data.reader.get_meta_data()
fps = round(video_meta_data['fps'])
duration = round(video_meta_data['duration'])
print(f"orginal fps: {fps} duration: {duration}")
assert start < duration and start + length < duration
if length == -1:
input_video = [video[i] for i in range(start*fps, (duration-start)*fps)]
input_video = [video[i] for i in range(start*fps, len(video))]
else:
input_video = [video[i] for i in range(start*fps, (start+length)*fps)]
print(f"{len(input_video)} frame will be to shade")
# Toon shading (20G VRAM)
torch.manual_seed(seed)
output_video = pipe(
@@ -241,7 +244,7 @@ class VideoShadeNode:
controlnet_frames=input_video, num_frames=len(input_video),
num_inference_steps=num_inference_steps, height=height, width=width,
animatediff_batch_size=animatediff_batch_size, animatediff_stride=animatediff_stride,
vram_limit_level=0,
vram_limit_level=vram_limit_level,
)
output_video = smoother(output_video)