From 9da6b79ecebc8c3ece409fc548612a683844c397 Mon Sep 17 00:00:00 2001
From: Bubbliiiing <47347516+bubbliiiing@users.noreply.github.com>
Date: Mon, 11 Nov 2024 13:48:53 +0800
Subject: [PATCH] Update Prompt && Update Readme && Fix control bug (#131)
* Update Readme
* Update prompt tips
* Fix bug in ui
* Fix control bug and Update readme
* Update prompt again
---
README.md | 42 ++-
README_zh-CN.md | 42 ++-
comfyui/comfyui_nodes.py | 16 +-
comfyui/v5/easyanimatev5_workflow_i2v.json | 102 +++++---
comfyui/v5/easyanimatev5_workflow_t2v.json | 68 +++--
comfyui/v5/easyanimatev5_workflow_v2v.json | 46 +++-
.../easyanimatev5_workflow_v2v_control.json | 46 +++-
easyanimate/api/api.py | 2 +-
easyanimate/ui/ui.py | 246 +++++++++++-------
predict_i2v.py | 9 +-
predict_t2v.py | 14 +-
predict_v2v.py | 12 +-
predict_v2v_control.py | 9 +-
13 files changed, 445 insertions(+), 209 deletions(-)
diff --git a/README.md b/README.md
index 91b2d90..e9b9e74 100644
--- a/README.md
+++ b/README.md
@@ -131,8 +131,7 @@ The results displayed are all based on image.
### EasyAnimateV5-12b-zh-InP
-Resolution-1024
-
+#### I2V
|
@@ -151,8 +150,6 @@ Resolution-1024
|
-Resolution-768
-
|
@@ -170,8 +167,6 @@ Resolution-768
|
-Resolution-512
-
|
@@ -189,6 +184,41 @@ Resolution-512
|
+#### T2V
+
+
+ |
+
+ |
+
+
+ |
+
+
+ |
+
+
+ |
+
+
+
+
+
+ |
+
+ |
+
+
+ |
+
+
+ |
+
+
+ |
+
+
+
### EasyAnimateV5-12b-zh-Control
diff --git a/README_zh-CN.md b/README_zh-CN.md
index 2d82f84..3a7e973 100644
--- a/README_zh-CN.md
+++ b/README_zh-CN.md
@@ -129,8 +129,7 @@ EasyAnimateV5:
### EasyAnimateV5-12b-zh-InP
-Resolution-1024
-
+#### I2V
|
@@ -149,8 +148,6 @@ Resolution-1024
|
-Resolution-768
-
|
@@ -168,8 +165,6 @@ Resolution-768
|
-Resolution-512
-
|
@@ -187,6 +182,41 @@ Resolution-512
|
+#### T2V
+
+
+ |
+
+ |
+
+
+ |
+
+
+ |
+
+
+ |
+
+
+
+
+
+ |
+
+ |
+
+
+ |
+
+
+ |
+
+
+ |
+
+
+
### EasyAnimateV5-12b-zh-Control
diff --git a/comfyui/comfyui_nodes.py b/comfyui/comfyui_nodes.py
index c882563..136b313 100644
--- a/comfyui/comfyui_nodes.py
+++ b/comfyui/comfyui_nodes.py
@@ -265,12 +265,16 @@ class LoadEasyAnimateModel:
)
else:
pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Control.from_pretrained(
- model_name,
- vae=vae,
- transformer=transformer,
- scheduler=scheduler,
- torch_dtype=weight_dtype
- )
+ model_name,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=vae,
+ transformer=transformer,
+ scheduler=scheduler,
+ torch_dtype=weight_dtype
+ )
if GPU_memory_mode == "sequential_cpu_offload":
pipeline.enable_sequential_cpu_offload()
elif GPU_memory_mode == "model_cpu_offload_and_qfloat8":
diff --git a/comfyui/v5/easyanimatev5_workflow_i2v.json b/comfyui/v5/easyanimatev5_workflow_i2v.json
index 2d43c94..0948a38 100644
--- a/comfyui/v5/easyanimatev5_workflow_i2v.json
+++ b/comfyui/v5/easyanimatev5_workflow_i2v.json
@@ -1,5 +1,5 @@
{
- "last_node_id": 83,
+ "last_node_id": 84,
"last_link_id": 48,
"nodes": [
{
@@ -52,31 +52,6 @@
"color": "#432",
"bgcolor": "#653"
},
- {
- "id": 78,
- "type": "Note",
- "pos": {
- "0": 18,
- "1": -46
- },
- "size": {
- "0": 210,
- "1": 58
- },
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [],
- "outputs": [],
- "properties": {
- "text": ""
- },
- "widgets_values": [
- "You can write prompt here\n(你可以在此填写提示词)"
- ],
- "color": "#432",
- "bgcolor": "#653"
- },
{
"id": 17,
"type": "VHS_VideoCombine",
@@ -86,10 +61,10 @@
},
"size": [
390.9534912109375,
- 535.9734235491071
+ 310
],
"flags": {},
- "order": 8,
+ "order": 9,
"mode": 0,
"inputs": [
{
@@ -97,7 +72,8 @@
"type": "IMAGE",
"link": 42,
"slot_index": 0,
- "label": "图像"
+ "label": "图像",
+ "shape": 7
},
{
"name": "audio",
@@ -168,7 +144,7 @@
"1": 282
},
"flags": {},
- "order": 7,
+ "order": 8,
"mode": 0,
"inputs": [
{
@@ -235,7 +211,7 @@
"1": 154
},
"flags": {},
- "order": 3,
+ "order": 2,
"mode": 0,
"inputs": [],
"outputs": [
@@ -272,7 +248,7 @@
"1": 314
},
"flags": {},
- "order": 4,
+ "order": 3,
"mode": 0,
"inputs": [],
"outputs": [
@@ -315,7 +291,7 @@
"1": 156.71620178222656
},
"flags": {},
- "order": 5,
+ "order": 4,
"mode": 0,
"inputs": [],
"outputs": [
@@ -349,7 +325,7 @@
"1": 183.83506774902344
},
"flags": {},
- "order": 6,
+ "order": 5,
"mode": 0,
"inputs": [],
"outputs": [
@@ -368,8 +344,58 @@
"Node name for S&R": "TextBox"
},
"widgets_values": [
- "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+ "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
+ },
+ {
+ "id": 78,
+ "type": "Note",
+ "pos": {
+ "0": 18,
+ "1": -46
+ },
+ "size": {
+ "0": 210,
+ "1": 58
+ },
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {
+ "text": ""
+ },
+ "widgets_values": [
+ "You can write prompt here\n(你可以在此填写提示词)"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 84,
+ "type": "Note",
+ "pos": {
+ "0": -98,
+ "1": 198
+ },
+ "size": [
+ 326.1556114026207,
+ 145.20905264909447
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {
+ "text": ""
+ },
+ "widgets_values": [
+ "Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
}
],
"links": [
@@ -455,10 +481,10 @@
"config": {},
"extra": {
"ds": {
- "scale": 0.7247295000000005,
+ "scale": 0.7513148009015777,
"offset": [
- 395.9831007505868,
- 450.6554192121478
+ 206.91128312862938,
+ 440.0942364134056
]
},
"workspace_info": {
diff --git a/comfyui/v5/easyanimatev5_workflow_t2v.json b/comfyui/v5/easyanimatev5_workflow_t2v.json
index c6f86a5..93a100f 100644
--- a/comfyui/v5/easyanimatev5_workflow_t2v.json
+++ b/comfyui/v5/easyanimatev5_workflow_t2v.json
@@ -1,5 +1,5 @@
{
- "last_node_id": 88,
+ "last_node_id": 89,
"last_link_id": 53,
"nodes": [
{
@@ -101,7 +101,7 @@
"1": 290
},
"flags": {},
- "order": 5,
+ "order": 6,
"mode": 0,
"inputs": [
{
@@ -214,7 +214,7 @@
"Node name for S&R": "TextBox"
},
"widgets_values": [
- "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+ "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
@@ -226,10 +226,10 @@
},
"size": [
390.9534912109375,
- 535.9734235491071
+ 310
],
"flags": {},
- "order": 6,
+ "order": 7,
"mode": 0,
"inputs": [
{
@@ -237,7 +237,8 @@
"type": "IMAGE",
"link": 50,
"slot_index": 0,
- "label": "图像"
+ "label": "图像",
+ "shape": 7
},
{
"name": "audio",
@@ -295,6 +296,31 @@
}
}
}
+ },
+ {
+ "id": 89,
+ "type": "Note",
+ "pos": {
+ "0": -97,
+ "1": 193
+ },
+ "size": [
+ 326.1556114026207,
+ 145.20905264909447
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {
+ "text": ""
+ },
+ "widgets_values": [
+ "Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
}
],
"links": [
@@ -332,18 +358,6 @@
]
],
"groups": [
- {
- "title": "Prompts",
- "bounding": [
- 218,
- -127,
- 450,
- 483
- ],
- "color": "#3f789e",
- "font_size": 24,
- "flags": {}
- },
{
"title": "Load EasyAnimate",
"bounding": [
@@ -355,15 +369,27 @@
"color": "#b06634",
"font_size": 24,
"flags": {}
+ },
+ {
+ "title": "Prompts",
+ "bounding": [
+ 218,
+ -127,
+ 450,
+ 483
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
}
],
"config": {},
"extra": {
"ds": {
- "scale": 0.8769226950000006,
+ "scale": 0.7513148009015777,
"offset": [
- 89.91314837012095,
- 505.1888927984372
+ 206.91128312862938,
+ 440.0942364134056
]
},
"workspace_info": {
diff --git a/comfyui/v5/easyanimatev5_workflow_v2v.json b/comfyui/v5/easyanimatev5_workflow_v2v.json
index 063c6a0..67d3121 100644
--- a/comfyui/v5/easyanimatev5_workflow_v2v.json
+++ b/comfyui/v5/easyanimatev5_workflow_v2v.json
@@ -1,5 +1,5 @@
{
- "last_node_id": 87,
+ "last_node_id": 88,
"last_link_id": 52,
"nodes": [
{
@@ -145,7 +145,7 @@
"Node name for S&R": "TextBox"
},
"widgets_values": [
- "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+ "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
@@ -194,7 +194,7 @@
"1": 306
},
"flags": {},
- "order": 7,
+ "order": 8,
"mode": 0,
"inputs": [
{
@@ -262,10 +262,10 @@
},
"size": [
390.9534912109375,
- 535.9734235491071
+ 310
],
"flags": {},
- "order": 8,
+ "order": 9,
"mode": 0,
"inputs": [
{
@@ -273,7 +273,8 @@
"type": "IMAGE",
"link": 48,
"slot_index": 0,
- "label": "图像"
+ "label": "图像",
+ "shape": 7
},
{
"name": "audio",
@@ -341,7 +342,7 @@
},
"size": [
252.056640625,
- 408.6037946428571
+ 262
],
"flags": {},
"order": 6,
@@ -416,6 +417,31 @@
}
}
}
+ },
+ {
+ "id": 88,
+ "type": "Note",
+ "pos": {
+ "0": -97,
+ "1": 195
+ },
+ "size": [
+ 326.1556091308594,
+ 145.20904541015625
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {
+ "text": ""
+ },
+ "widgets_values": [
+ "Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
}
],
"links": [
@@ -501,10 +527,10 @@
"config": {},
"extra": {
"ds": {
- "scale": 0.6588450000000011,
+ "scale": 0.7513148009015777,
"offset": [
- 599.4920942053536,
- 496.29399099271234
+ 206.91128312862938,
+ 440.0942364134056
]
},
"workspace_info": {
diff --git a/comfyui/v5/easyanimatev5_workflow_v2v_control.json b/comfyui/v5/easyanimatev5_workflow_v2v_control.json
index f4ba2b4..c5caa61 100644
--- a/comfyui/v5/easyanimatev5_workflow_v2v_control.json
+++ b/comfyui/v5/easyanimatev5_workflow_v2v_control.json
@@ -1,5 +1,5 @@
{
- "last_node_id": 87,
+ "last_node_id": 88,
"last_link_id": 53,
"nodes": [
{
@@ -108,7 +108,7 @@
"Node name for S&R": "TextBox"
},
"widgets_values": [
- "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+ "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
@@ -120,10 +120,10 @@
},
"size": [
390.9534912109375,
- 973.1686096191406
+ 310
],
"flags": {},
- "order": 8,
+ "order": 9,
"mode": 0,
"inputs": [
{
@@ -131,7 +131,8 @@
"type": "IMAGE",
"link": 48,
"slot_index": 0,
- "label": "图像"
+ "label": "图像",
+ "shape": 7
},
{
"name": "audio",
@@ -236,7 +237,7 @@
},
"size": [
252.056640625,
- 688.5451388888889
+ 262
],
"flags": {},
"order": 5,
@@ -358,7 +359,7 @@
"1": 306
},
"flags": {},
- "order": 7,
+ "order": 8,
"mode": 0,
"inputs": [
{
@@ -416,6 +417,31 @@
1,
"DDIM"
]
+ },
+ {
+ "id": 88,
+ "type": "Note",
+ "pos": {
+ "0": -99,
+ "1": 197
+ },
+ "size": [
+ 326.1556114026207,
+ 145.20905264909447
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {
+ "text": ""
+ },
+ "widgets_values": [
+ "Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
}
],
"links": [
@@ -501,10 +527,10 @@
"config": {},
"extra": {
"ds": {
- "scale": 0.7247295000000012,
+ "scale": 0.7513148009015777,
"offset": [
- 302.3321359435954,
- 322.8143100800612
+ 206.91128312862938,
+ 440.0942364134056
]
},
"workspace_info": {
diff --git a/easyanimate/api/api.py b/easyanimate/api/api.py
index c604151..f278352 100644
--- a/easyanimate/api/api.py
+++ b/easyanimate/api/api.py
@@ -93,7 +93,7 @@ def infer_forward_api(_: gr.Blocks, app: FastAPI, controller):
lora_model_path = datas.get('lora_model_path', 'none')
lora_alpha_slider = datas.get('lora_alpha_slider', 0.55)
prompt_textbox = datas.get('prompt_textbox', None)
- negative_prompt_textbox = datas.get('negative_prompt_textbox', 'Blurring, mutation, deformation, distortion, dark and solid, comics.')
+ negative_prompt_textbox = datas.get('negative_prompt_textbox', 'Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art.')
sampler_dropdown = datas.get('sampler_dropdown', 'Euler')
sample_step_slider = datas.get('sample_step_slider', 30)
resize_method = datas.get('resize_method', "Generate by")
diff --git a/easyanimate/ui/ui.py b/easyanimate/ui/ui.py
index 9783abf..f6736af 100644
--- a/easyanimate/ui/ui.py
+++ b/easyanimate/ui/ui.py
@@ -39,6 +39,8 @@ from easyanimate.pipeline.pipeline_easyanimate_multi_text_encoder import \
EasyAnimatePipeline_Multi_Text_Encoder
from easyanimate.pipeline.pipeline_easyanimate_multi_text_encoder_inpaint import \
EasyAnimatePipeline_Multi_Text_Encoder_Inpaint
+from easyanimate.pipeline.pipeline_easyanimate_multi_text_encoder_control import \
+ EasyAnimatePipeline_Multi_Text_Encoder_Control
from easyanimate.utils.lora_utils import merge_lora, unmerge_lora
from easyanimate.utils.utils import (
get_image_to_video_latent, get_video_to_video_latent,
@@ -225,56 +227,69 @@ class EasyAnimateController:
subfolder="scheduler"
)
- if self.inference_config['text_encoder_kwargs'].get('enable_multi_text_encoder', False):
- if self.transformer.config.in_channels != self.vae.config.latent_channels:
- self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Inpaint.from_pretrained(
- diffusion_transformer_dropdown,
- text_encoder=text_encoder,
- text_encoder_2=text_encoder_2,
- tokenizer=tokenizer,
- tokenizer_2=tokenizer_2,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype,
- clip_image_encoder=clip_image_encoder,
- clip_image_processor=clip_image_processor,
- )
+ if self.model_type == "Inpaint":
+ if self.inference_config['text_encoder_kwargs'].get('enable_multi_text_encoder', False):
+ if self.transformer.config.in_channels != self.vae.config.latent_channels:
+ self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Inpaint.from_pretrained(
+ diffusion_transformer_dropdown,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype,
+ clip_image_encoder=clip_image_encoder,
+ clip_image_processor=clip_image_processor,
+ )
+ else:
+ self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder.from_pretrained(
+ diffusion_transformer_dropdown,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype
+ )
else:
- self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder.from_pretrained(
- diffusion_transformer_dropdown,
- text_encoder=text_encoder,
- text_encoder_2=text_encoder_2,
- tokenizer=tokenizer,
- tokenizer_2=tokenizer_2,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype
- )
+ if self.transformer.config.in_channels != self.vae.config.latent_channels:
+ self.pipeline = EasyAnimateInpaintPipeline(
+ diffusion_transformer_dropdown,
+ text_encoder=text_encoder,
+ tokenizer=tokenizer,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype,
+ clip_image_encoder=clip_image_encoder,
+ clip_image_processor=clip_image_processor,
+ )
+ else:
+ self.pipeline = EasyAnimatePipeline(
+ diffusion_transformer_dropdown,
+ text_encoder=text_encoder,
+ tokenizer=tokenizer,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype
+ )
else:
- if self.transformer.config.in_channels != self.vae.config.latent_channels:
- self.pipeline = EasyAnimateInpaintPipeline(
- diffusion_transformer_dropdown,
- text_encoder=text_encoder,
- tokenizer=tokenizer,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype,
- clip_image_encoder=clip_image_encoder,
- clip_image_processor=clip_image_processor,
- )
- else:
- self.pipeline = EasyAnimatePipeline(
- diffusion_transformer_dropdown,
- text_encoder=text_encoder,
- tokenizer=tokenizer,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype
- )
+ self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Control.from_pretrained(
+ diffusion_transformer_dropdown,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype
+ )
if self.GPU_memory_mode == "sequential_cpu_offload":
self.pipeline.enable_sequential_cpu_offload()
@@ -283,7 +298,7 @@ class EasyAnimateController:
self.pipeline.enable_autocast_float8_transformer()
convert_weight_dtype_wrapper(self.pipeline.transformer, self.weight_dtype)
else:
- self.GPU_memory_mode.enable_model_cpu_offload()
+ self.pipeline.enable_model_cpu_offload()
print("Update diffusion transformer done")
return gr.update()
@@ -752,7 +767,13 @@ def ui(GPU_memory_mode, weight_dtype):
)
prompt_textbox = gr.Textbox(label="Prompt (正向提示词)", lines=2, value="A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.")
- negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="Blurring, mutation, deformation, distortion, dark and solid, comics." )
+ gr.Markdown(
+ """
+ Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability. Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
+ 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性。在neg prompt中添加"安静,固定"等词语可以增加动态性。
+ """
+ )
+ negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code.. " )
with gr.Row():
with gr.Column():
@@ -1058,56 +1079,69 @@ class EasyAnimateController_Modelscope:
subfolder="scheduler"
)
- if self.inference_config['text_encoder_kwargs'].get('enable_multi_text_encoder', False):
- if self.transformer.config.in_channels != self.vae.config.latent_channels:
- self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Inpaint.from_pretrained(
- model_name,
- text_encoder=text_encoder,
- text_encoder_2=text_encoder_2,
- tokenizer=tokenizer,
- tokenizer_2=tokenizer_2,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype,
- clip_image_encoder=clip_image_encoder,
- clip_image_processor=clip_image_processor,
- )
+ if model_type == "Inpaint":
+ if self.inference_config['text_encoder_kwargs'].get('enable_multi_text_encoder', False):
+ if self.transformer.config.in_channels != self.vae.config.latent_channels:
+ self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Inpaint.from_pretrained(
+ model_name,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype,
+ clip_image_encoder=clip_image_encoder,
+ clip_image_processor=clip_image_processor,
+ )
+ else:
+ self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder.from_pretrained(
+ model_name,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype
+ )
else:
- self.pipeline = EasyAnimatePipeline_Multi_Text_Encoder.from_pretrained(
- model_name,
- text_encoder=text_encoder,
- text_encoder_2=text_encoder_2,
- tokenizer=tokenizer,
- tokenizer_2=tokenizer_2,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype
- )
+ if self.transformer.config.in_channels != self.vae.config.latent_channels:
+ self.pipeline = EasyAnimateInpaintPipeline(
+ model_name,
+ text_encoder=text_encoder,
+ tokenizer=tokenizer,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype,
+ clip_image_encoder=clip_image_encoder,
+ clip_image_processor=clip_image_processor,
+ )
+ else:
+ self.pipeline = EasyAnimatePipeline(
+ model_name,
+ text_encoder=text_encoder,
+ tokenizer=tokenizer,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=self.weight_dtype
+ )
else:
- if self.transformer.config.in_channels != self.vae.config.latent_channels:
- self.pipeline = EasyAnimateInpaintPipeline(
- model_name,
- text_encoder=text_encoder,
- tokenizer=tokenizer,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype,
- clip_image_encoder=clip_image_encoder,
- clip_image_processor=clip_image_processor,
- )
- else:
- self.pipeline = EasyAnimatePipeline(
- model_name,
- text_encoder=text_encoder,
- tokenizer=tokenizer,
- vae=self.vae,
- transformer=self.transformer,
- scheduler=scheduler,
- torch_dtype=self.weight_dtype
- )
+ pipeline = EasyAnimatePipeline_Multi_Text_Encoder_Control.from_pretrained(
+ model_name,
+ text_encoder=text_encoder,
+ text_encoder_2=text_encoder_2,
+ tokenizer=tokenizer,
+ tokenizer_2=tokenizer_2,
+ vae=self.vae,
+ transformer=self.transformer,
+ scheduler=scheduler,
+ torch_dtype=weight_dtype
+ )
if GPU_memory_mode == "sequential_cpu_offload":
self.pipeline.enable_sequential_cpu_offload()
@@ -1407,7 +1441,13 @@ def ui_modelscope(model_type, edition, config_path, model_name, savedir_sample,
)
prompt_textbox = gr.Textbox(label="Prompt (正向提示词)", lines=2, value="A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.")
- negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="Blurring, mutation, deformation, distortion, dark and solid, comics." )
+ gr.Markdown(
+ """
+ Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability. Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
+ 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性。在neg prompt中添加"安静,固定"等词语可以增加动态性。
+ """
+ )
+ negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code.. " )
with gr.Row():
with gr.Column():
@@ -1821,7 +1861,13 @@ def ui_eas(edition, config_path, model_name, savedir_sample):
)
prompt_textbox = gr.Textbox(label="Prompt", lines=2, value="A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.")
- negative_prompt_textbox = gr.Textbox(label="Negative prompt", lines=2, value="Blurring, mutation, deformation, distortion, dark and solid, comics." )
+ gr.Markdown(
+ """
+ Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability. Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
+ 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性。在neg prompt中添加"安静,固定"等词语可以增加动态性。
+ """
+ )
+ negative_prompt_textbox = gr.Textbox(label="Negative prompt", lines=2, value="Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code.. " )
with gr.Row():
with gr.Column():
diff --git a/predict_i2v.py b/predict_i2v.py
index b1b0c50..b8deb03 100644
--- a/predict_i2v.py
+++ b/predict_i2v.py
@@ -70,10 +70,15 @@ validation_image_end = None
# EasyAnimateV1, V2 and V3 support English.
# EasyAnimateV4 and V5 support English and Chinese.
+# 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性
+# 在neg prompt中添加"安静,固定"等词语可以增加动态性。
prompt = "一条狗正在摇头。质量高、杰作、最佳品质、高分辨率、超精细、梦幻般。"
-negative_prompt = "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
+#
+# Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability
+# Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
# prompt = "The dog is shaking head. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
-# negative_prompt = "Blurring, mutation, deformation, distortion, dark and solid, comics."
+# negative_prompt = "Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code."
guidance_scale = 6.0
seed = 43
num_inference_steps = 50
diff --git a/predict_t2v.py b/predict_t2v.py
index 0cbc87e..2b8522b 100644
--- a/predict_t2v.py
+++ b/predict_t2v.py
@@ -63,12 +63,18 @@ fps = 8
# Use torch.float16 if GPU does not support torch.bfloat16
# ome graphics cards, such as v100, 2080ti, do not support torch.bfloat16
weight_dtype = torch.bfloat16
-# # EasyAnimateV1, V2 and V3 support English.
+
+# EasyAnimateV1, V2 and V3 support English.
# EasyAnimateV4 and V5 support English and Chinese.
+# 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性
+# 在neg prompt中添加"安静,固定"等词语可以增加动态性。
prompt = "一条狗正在摇头。质量高、杰作、最佳品质、高分辨率、超精细、梦幻般。"
-negative_prompt = "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
-# prompt = "A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
-# negative_prompt = "Blurring, mutation, deformation, distortion, dark and solid, comics."
+negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
+#
+# Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability
+# Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
+# prompt = "The dog is shaking head. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
+# negative_prompt = "Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code."
guidance_scale = 6.0
seed = 43
num_inference_steps = 50
diff --git a/predict_v2v.py b/predict_v2v.py
index 24e8fa2..056d38f 100644
--- a/predict_v2v.py
+++ b/predict_v2v.py
@@ -65,13 +65,17 @@ weight_dtype = torch.bfloat16
validation_video = "asset/1.mp4"
denoise_strength = 0.70
-# prompts
-# # EasyAnimateV1, V2 and V3 support English.
+# EasyAnimateV1, V2 and V3 support English.
# EasyAnimateV4 and V5 support English and Chinese.
+# 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性
+# 在neg prompt中添加"安静,固定"等词语可以增加动态性。
prompt = "一只猫正在弹吉他。"
-negative_prompt = "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
+#
+# Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability
+# Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
# prompt = "A cute cat is playing the guitar. "
-# negative_prompt = "Blurring, mutation, deformation, distortion, dark and solid, comics."
+# negative_prompt = "Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code.. "
guidance_scale = 6.0
seed = 43
num_inference_steps = 50
diff --git a/predict_v2v_control.py b/predict_v2v_control.py
index 84e1c02..adae034 100644
--- a/predict_v2v_control.py
+++ b/predict_v2v_control.py
@@ -59,8 +59,15 @@ control_video = "asset/pose.mp4"
# EasyAnimateV1, V2 and V3 support English.
# EasyAnimateV4 and V5 support English and Chinese.
+# 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性
+# 在neg prompt中添加"安静,固定"等词语可以增加动态性。
prompt = "一位年轻女子,有着美丽清澈的眼睛和金发,穿着白色的衣服在扭动身体,相机聚焦在她的脸上。质量高、杰作、最佳品质、高分辨率、超精细、梦幻般。"
-negative_prompt = "模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。"
+negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
+#
+# Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability
+# Adding words such as "quiet, solid" to the neg prompt can increase dynamism.
+# prompt = "A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
+# negative_prompt = "Twisted body, limb deformities, text captions, comic, static, ugly, error, messy code."
guidance_scale = 6.0
seed = 43
num_inference_steps = 50