more refactoring

This commit is contained in:
kijai
2024-06-02 18:57:23 +03:00
parent 86e789d35d
commit df0ade0778
10 changed files with 2118 additions and 881 deletions
+6 -6
View File
@@ -55,7 +55,7 @@ model:
fs_condition: true
first_stage_config:
target: .lvdm.models.autoencoder.AutoencoderKL
target: .lvdm.models.autoencoder_old.AutoencoderKL
params:
embed_dim: 4
monitor: val/rec_loss
@@ -77,11 +77,11 @@ model:
lossconfig:
target: torch.nn.Identity
cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
params:
freeze: true
layer: "penultimate"
# cond_stage_config:
# target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
# params:
# freeze: true
# layer: "penultimate"
img_cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPImageEmbedderV2
+6 -6
View File
@@ -50,7 +50,7 @@ model:
fs_condition: true
first_stage_config:
target: .lvdm.models.autoencoder.AutoencoderKL
target: .lvdm.models.autoencoder_old.AutoencoderKL
params:
embed_dim: 4
monitor: val/rec_loss
@@ -72,11 +72,11 @@ model:
lossconfig:
target: torch.nn.Identity
cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
params:
freeze: true
layer: "penultimate"
# cond_stage_config:
# target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
# params:
# freeze: true
# layer: "penultimate"
img_cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPImageEmbedderV2
+6 -6
View File
@@ -55,7 +55,7 @@ model:
fs_condition: true
first_stage_config:
target: .lvdm.models.autoencoder.AutoencoderKL
target: .lvdm.models.autoencoder_old.AutoencoderKL
params:
embed_dim: 4
monitor: val/rec_loss
@@ -77,11 +77,11 @@ model:
lossconfig:
target: torch.nn.Identity
cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
params:
freeze: true
layer: "penultimate"
# cond_stage_config:
# target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
# params:
# freeze: true
# layer: "penultimate"
img_cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPImageEmbedderV2
+6 -6
View File
@@ -55,7 +55,7 @@ model:
fs_condition: true
first_stage_config:
target: .lvdm.models.autoencoder.AutoencoderKL
target: .lvdm.models.autoencoder_old.AutoencoderKL
params:
embed_dim: 4
monitor: val/rec_loss
@@ -77,11 +77,11 @@ model:
lossconfig:
target: torch.nn.Identity
cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
params:
freeze: true
layer: "penultimate"
# cond_stage_config:
# target: .lvdm.modules.encoders.condition.FrozenOpenCLIPEmbedder
# params:
# freeze: true
# layer: "penultimate"
img_cond_stage_config:
target: .lvdm.modules.encoders.condition.FrozenOpenCLIPImageEmbedderV2
+598
View File
@@ -0,0 +1,598 @@
{
"last_node_id": 60,
"last_link_id": 148,
"nodes": [
{
"id": 6,
"type": "GetImageSizeAndCount",
"pos": [
1420,
269
],
"size": {
"0": 210,
"1": 86
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 145
}
],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": [
143
],
"shape": 3,
"slot_index": 0
},
{
"name": "896 width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "512 height",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "1 count",
"type": "INT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "GetImageSizeAndCount"
}
},
{
"id": 5,
"type": "ImageResizeKJ",
"pos": [
861,
197
],
"size": {
"0": 315,
"1": 242
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 2
},
{
"name": "get_image_size",
"type": "IMAGE",
"link": null
},
{
"name": "width_input",
"type": "INT",
"link": null,
"widget": {
"name": "width_input"
}
},
{
"name": "height_input",
"type": "INT",
"link": null,
"widget": {
"name": "height_input"
}
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
145
],
"shape": 3,
"slot_index": 0
},
{
"name": "width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "height",
"type": "INT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "ImageResizeKJ"
},
"widgets_values": [
1024,
512,
"lanczos",
true,
64,
0,
0
]
},
{
"id": 58,
"type": "DynamiCrafterI2V",
"pos": [
1759,
160
],
"size": {
"0": 315,
"1": 418
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "DCMODEL",
"link": 138
},
{
"name": "clip_vision",
"type": "CLIP_VISION",
"link": 146
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 140
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 141
},
{
"name": "image",
"type": "IMAGE",
"link": 143
},
{
"name": "image2",
"type": "IMAGE",
"link": null
},
{
"name": "mask",
"type": "MASK",
"link": null
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
142
],
"shape": 3,
"slot_index": 0
},
{
"name": "last_image",
"type": "IMAGE",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "DynamiCrafterI2V"
},
"widgets_values": [
20,
7,
1,
16,
619731667089947,
"fixed",
10,
true,
"auto",
16,
4
]
},
{
"id": 50,
"type": "CLIPTextEncode",
"pos": [
1206,
777
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 148
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
141
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
""
]
},
{
"id": 1,
"type": "LoadImage",
"pos": [
490,
200
],
"size": {
"0": 315,
"1": 314
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
2
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": [],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"40.png",
"image"
]
},
{
"id": 60,
"type": "DownloadAndLoadCLIPModel",
"pos": [
728,
647
],
"size": [
371.0226473136131,
64.01405022360109
],
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "clip",
"type": "CLIP",
"links": [
147,
148
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadCLIPModel"
},
"widgets_values": [
"stable-diffusion-2-1-clip-fp16.safetensors"
]
},
{
"id": 49,
"type": "CLIPTextEncode",
"pos": [
1204,
522
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 147,
"slot_index": 0
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
140
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"branches in wind"
]
},
{
"id": 59,
"type": "DownloadAndLoadCLIPVisionModel",
"pos": [
1207,
52
],
"size": {
"0": 315,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "clip_vision",
"type": "CLIP_VISION",
"links": [
146
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadCLIPVisionModel"
},
"widgets_values": [
"CLIP-ViT-H-fp16.safetensors"
]
},
{
"id": 52,
"type": "DownloadAndLoadDynamiCrafterModel",
"pos": [
1209,
-109
],
"size": {
"0": 389.78204345703125,
"1": 106
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "DynCraft_model",
"type": "DCMODEL",
"links": [
138
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadDynamiCrafterModel"
},
"widgets_values": [
"dynamicrafter_1024_v1_bf16.safetensors",
"auto",
true
]
},
{
"id": 29,
"type": "VHS_VideoCombine",
"pos": [
2099,
-128
],
"size": [
1270,
1018.2857142857143
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 142
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "AnimateDiff",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"pingpong": false,
"save_output": false,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "AnimateDiff_00005.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4"
}
}
}
}
],
"links": [
[
2,
1,
0,
5,
0,
"IMAGE"
],
[
138,
52,
0,
58,
0,
"DCMODEL"
],
[
140,
49,
0,
58,
2,
"CONDITIONING"
],
[
141,
50,
0,
58,
3,
"CONDITIONING"
],
[
142,
58,
0,
29,
0,
"IMAGE"
],
[
143,
6,
0,
58,
4,
"IMAGE"
],
[
145,
5,
0,
6,
0,
"IMAGE"
],
[
146,
59,
0,
58,
1,
"CLIP_VISION"
],
[
147,
60,
0,
49,
0,
"CLIP"
],
[
148,
60,
0,
50,
0,
"CLIP"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.6830134553650712,
"offset": [
-264.2064939933011,
412.92671373513866
]
}
},
"version": 0.4
}
@@ -1,431 +0,0 @@
{
"last_node_id": 101,
"last_link_id": 194,
"nodes": [
{
"id": 72,
"type": "DynamiCrafterModelLoader",
"pos": [
798,
70
],
"size": [
402.70587216796844,
82
],
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "DynCraft_model",
"type": "DCMODEL",
"links": [
181
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DynamiCrafterModelLoader"
},
"widgets_values": [
"dynamicrafter\\dynamicrafter_512_interp_v1.ckpt",
"fp16"
]
},
{
"id": 71,
"type": "LoadImage",
"pos": [
354,
71
],
"size": {
"0": 409.56280517578125,
"1": 355.5024719238281
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
183
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"40.png",
"image"
]
},
{
"id": 100,
"type": "GetImageRangeFromBatch",
"pos": [
884,
614
],
"size": {
"0": 315,
"1": 82
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 191
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
193
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "GetImageRangeFromBatch"
},
"widgets_values": [
0,
15
]
},
{
"id": 101,
"type": "RIFE VFI",
"pos": [
789,
746
],
"size": {
"0": 443.4000244140625,
"1": 198
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "frames",
"type": "IMAGE",
"link": 193
},
{
"name": "optional_interpolation_states",
"type": "INTERPOLATION_STATES",
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
194
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "RIFE VFI"
},
"widgets_values": [
"rife49.pth",
10,
3,
true,
true,
1
]
},
{
"id": 76,
"type": "VHS_VideoCombine",
"pos": [
1285,
63
],
"size": [
941.1651000976562,
822.1553688049316
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 194,
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "batch_manager",
"type": "VHS_BatchManager",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 20,
"loop_count": 0,
"filename_prefix": "DynamiCrafter",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "DynamiCrafter_00077.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4"
}
}
}
},
{
"id": 96,
"type": "DynamiCrafterI2V",
"pos": [
804,
207
],
"size": {
"0": 400,
"1": 352
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "DCMODEL",
"link": 181
},
{
"name": "image",
"type": "IMAGE",
"link": 184,
"slot_index": 1
},
{
"name": "image2",
"type": "IMAGE",
"link": 186
},
{
"name": "mask",
"type": "MASK",
"link": null
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
191
],
"shape": 3,
"slot_index": 0
},
{
"name": "last_image",
"type": "IMAGE",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "DynamiCrafterI2V"
},
"widgets_values": [
40,
7,
1,
16,
"flowery branches moved by the wind",
710950197958318,
"fixed",
8,
true,
"auto"
]
},
{
"id": 90,
"type": "ImageResize+",
"pos": [
527,
479
],
"size": [
256.29377216796854,
218
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 183
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
169,
184,
186
],
"shape": 3,
"slot_index": 0
},
{
"name": "width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "height",
"type": "INT",
"links": null,
"shape": 3,
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "ImageResize+"
},
"widgets_values": [
512,
288,
"lanczos",
false,
"always",
0
]
}
],
"links": [
[
169,
90,
0,
93,
0,
"IMAGE"
],
[
181,
72,
0,
96,
0,
"DCMODEL"
],
[
183,
71,
0,
90,
0,
"IMAGE"
],
[
184,
90,
0,
96,
1,
"IMAGE"
],
[
186,
90,
0,
96,
2,
"IMAGE"
],
[
191,
96,
0,
100,
0,
"IMAGE"
],
[
193,
100,
0,
101,
0,
"IMAGE"
],
[
194,
101,
0,
76,
0,
"IMAGE"
]
],
"groups": [],
"config": {},
"extra": {},
"version": 0.4
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,913 @@
{
"last_node_id": 66,
"last_link_id": 151,
"nodes": [
{
"id": 2,
"type": "LoadImage",
"pos": [
486,
567
],
"size": {
"0": 315,
"1": 314
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
6
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"72109_125.mp4_00-00 (2).png",
"image"
]
},
{
"id": 1,
"type": "LoadImage",
"pos": [
490,
196
],
"size": {
"0": 315,
"1": 314
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
2
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": [],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"clipspace/clipspace-mask-8168.9000000003725.png [input]",
"image"
]
},
{
"id": 5,
"type": "ImageResizeKJ",
"pos": [
861,
197
],
"size": {
"0": 315,
"1": 242
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 2
},
{
"name": "get_image_size",
"type": "IMAGE",
"link": null
},
{
"name": "width_input",
"type": "INT",
"link": null,
"widget": {
"name": "width_input"
}
},
{
"name": "height_input",
"type": "INT",
"link": null,
"widget": {
"name": "height_input"
}
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
71,
73
],
"shape": 3,
"slot_index": 0
},
{
"name": "width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "height",
"type": "INT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "ImageResizeKJ"
},
"widgets_values": [
512,
512,
"lanczos",
true,
64,
0,
0
]
},
{
"id": 7,
"type": "ImageResizeKJ",
"pos": [
845,
504
],
"size": {
"0": 315,
"1": 242
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 6
},
{
"name": "get_image_size",
"type": "IMAGE",
"link": 73
},
{
"name": "width_input",
"type": "INT",
"link": null,
"widget": {
"name": "width_input"
}
},
{
"name": "height_input",
"type": "INT",
"link": null,
"widget": {
"name": "height_input"
}
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
128
],
"shape": 3,
"slot_index": 0
},
{
"name": "width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "height",
"type": "INT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "ImageResizeKJ"
},
"widgets_values": [
512,
512,
"lanczos",
true,
64,
0,
0
]
},
{
"id": 28,
"type": "ImageBatchMulti",
"pos": [
1405,
211
],
"size": {
"0": 210,
"1": 102
},
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "image_1",
"type": "IMAGE",
"link": 71
},
{
"name": "image_2",
"type": "IMAGE",
"link": 128
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
93
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "ImageBatchMulti"
},
"widgets_values": [
2,
null
]
},
{
"id": 6,
"type": "GetImageSizeAndCount",
"pos": [
1409,
362
],
"size": {
"0": 210,
"1": 86
},
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 93
}
],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": [
136
],
"shape": 3,
"slot_index": 0
},
{
"name": "512 width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "320 height",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "2 count",
"type": "INT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "GetImageSizeAndCount"
}
},
{
"id": 49,
"type": "CLIPTextEncode",
"pos": [
1317,
526
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 147,
"slot_index": 0
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
134
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"anime scene"
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 64,
"type": "Reroute",
"pos": [
1201,
703
],
"size": [
75,
26
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "",
"type": "*",
"link": 146
}
],
"outputs": [
{
"name": "CLIP",
"type": "CLIP",
"links": [
147,
148
],
"slot_index": 0
}
],
"properties": {
"showOutputText": true,
"horizontal": false
}
},
{
"id": 50,
"type": "CLIPTextEncode",
"pos": [
1322,
775
],
"size": [
400.4130416016717,
110.5309337152662
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 148
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
135
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
""
],
"color": "#322",
"bgcolor": "#533"
},
{
"id": 59,
"type": "DownloadAndLoadCLIPModel",
"pos": [
992,
12
],
"size": {
"0": 343.63671875,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "clip",
"type": "CLIP",
"links": [
146
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadCLIPModel"
},
"widgets_values": [
"stable-diffusion-2-1-clip-fp16.safetensors"
]
},
{
"id": 61,
"type": "DownloadAndLoadCLIPVisionModel",
"pos": [
992,
-100
],
"size": {
"0": 384.1668395996094,
"1": 58.00978088378906
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "clip_vision",
"type": "CLIP_VISION",
"links": [
145
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadCLIPVisionModel"
},
"widgets_values": [
"CLIP-ViT-H-fp16.safetensors"
]
},
{
"id": 52,
"type": "DownloadAndLoadDynamiCrafterModel",
"pos": [
991,
-260
],
"size": {
"0": 389.78204345703125,
"1": 106
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "DynCraft_model",
"type": "DCMODEL",
"links": [
132
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadDynamiCrafterModel"
},
"widgets_values": [
"tooncrafter_512_interp-fp16.safetensors",
"auto",
false
]
},
{
"id": 29,
"type": "VHS_VideoCombine",
"pos": [
2239,
206
],
"size": [
1271.3231201171875,
1086.0769500732422
],
"flags": {},
"order": 15,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 151
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "AnimateDiff",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"pingpong": false,
"save_output": false,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "AnimateDiff_00004.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4"
}
}
}
},
{
"id": 66,
"type": "VAELoader",
"pos": [
1834,
37
],
"size": [
379.34175114586014,
58
],
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "VAE",
"type": "VAE",
"links": [
149
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "VAELoader"
},
"widgets_values": [
"vae-ft-mse-840000-ema-pruned.safetensors"
]
},
{
"id": 65,
"type": "VAEDecode",
"pos": [
2255,
86
],
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 14,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 150,
"slot_index": 0
},
{
"name": "vae",
"type": "VAE",
"link": 149,
"slot_index": 1
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
151
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
}
},
{
"id": 57,
"type": "ToonCrafterInterpolation",
"pos": [
1850,
190
],
"size": {
"0": 315,
"1": 330
},
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "DCMODEL",
"link": 132
},
{
"name": "clip_vision",
"type": "CLIP_VISION",
"link": 145
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 134
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 135
},
{
"name": "images",
"type": "IMAGE",
"link": 136
}
],
"outputs": [
{
"name": "samples",
"type": "LATENT",
"links": [
150
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "ToonCrafterInterpolation"
},
"widgets_values": [
20,
7,
1,
16,
1,
"fixed",
10,
"auto",
1
]
}
],
"links": [
[
2,
1,
0,
5,
0,
"IMAGE"
],
[
6,
2,
0,
7,
0,
"IMAGE"
],
[
71,
5,
0,
28,
0,
"IMAGE"
],
[
73,
5,
0,
7,
1,
"IMAGE"
],
[
93,
28,
0,
6,
0,
"IMAGE"
],
[
128,
7,
0,
28,
1,
"IMAGE"
],
[
132,
52,
0,
57,
0,
"DCMODEL"
],
[
134,
49,
0,
57,
2,
"CONDITIONING"
],
[
135,
50,
0,
57,
3,
"CONDITIONING"
],
[
136,
6,
0,
57,
4,
"IMAGE"
],
[
145,
61,
0,
57,
1,
"CLIP_VISION"
],
[
146,
59,
0,
64,
0,
"*"
],
[
147,
64,
0,
49,
0,
"CLIP"
],
[
148,
64,
0,
50,
0,
"CLIP"
],
[
149,
66,
0,
65,
1,
"VAE"
],
[
150,
57,
0,
65,
0,
"LATENT"
],
[
151,
65,
0,
29,
0,
"IMAGE"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.6830134553650709,
"offset": [
-412.7737816939562,
411.4055448024345
]
}
},
"version": 0.4
}
+128 -64
View File
@@ -32,21 +32,6 @@ def convert_dtype(dtype_str):
script_directory = os.path.dirname(os.path.abspath(__file__))
class OpenCLIPVisionSelect:
@classmethod
def INPUT_TYPES(s):
return {"required": { "clip_name": (folder_paths.get_filename_list("clip_vision"), ),
}}
RETURN_TYPES = ("OPENCLIPVISIONPATH",)
FUNCTION = "getpath"
CATEGORY = "DynamiCrafterWrapper"
def getpath(self, clip_name):
clip_path = folder_paths.get_full_path("clip_vision", clip_name)
return (clip_path,)
class DownloadAndLoadDynamiCrafterModel:
@classmethod
def INPUT_TYPES(s):
@@ -138,6 +123,100 @@ class DownloadAndLoadDynamiCrafterModel:
print(f"Model using dtype: {self.model.dtype}")
return (self.model,)
class DownloadAndLoadCLIPModel:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"model": (
[ 'stable-diffusion-2-1-clip-fp16.safetensors',
'stable-diffusion-2-1-clip.safetensors',
],
{
"default": 'stable-diffusion-2-1-clip-fp16.safetensors'
}),
},
}
RETURN_TYPES = ("CLIP",)
RETURN_NAMES = ("clip",)
FUNCTION = "loadmodel"
CATEGORY = "DynamiCrafterWrapper"
def loadmodel(self, model):
import shutil
mm.soft_empty_cache()
download_path = os.path.join(folder_paths.models_dir, "temp")
model_path = os.path.join(folder_paths.models_dir, "clip", model)
if not os.path.exists(model_path):
print(f"Downloading model to: {model_path}")
filename = "model.fp16.safetensors" if "fp16" in model else "model.safetensors"
subfolder = "text_encoder"
from huggingface_hub import hf_hub_download
hf_hub_download(repo_id="stabilityai/stable-diffusion-2-1",
subfolder = subfolder,
filename = filename,
local_dir=download_path,
local_dir_use_symlinks=False)
source_file_path = os.path.join(download_path, subfolder, filename)
destination_file_path = model_path
shutil.move(source_file_path, destination_file_path)
clip_type = comfy.sd.CLIPType.STABLE_DIFFUSION
clip = comfy.sd.load_clip(ckpt_paths = [model_path], embedding_directory=folder_paths.get_folder_paths("embeddings"), clip_type=clip_type)
print(f"Loading model from: {model_path}")
return (clip,)
class DownloadAndLoadCLIPVisionModel:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"model": (
[ 'CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors',
'CLIP-ViT-H-fp16.safetensors',
],
{
"default": 'CLIP-ViT-H-fp16.safetensors'
}),
},
}
RETURN_TYPES = ("CLIP_VISION",)
RETURN_NAMES = ("clip_vision",)
FUNCTION = "loadmodel"
CATEGORY = "DynamiCrafterWrapper"
def loadmodel(self, model):
import shutil
mm.soft_empty_cache()
download_path = os.path.join(folder_paths.models_dir, "temp")
model_path = os.path.join(folder_paths.models_dir, "clip_vision", model)
if not os.path.exists(model_path):
print(f"Downloading model to: {model_path}")
from huggingface_hub import hf_hub_download
if "fp16" in model:
hf_hub_download(repo_id="Kijai/CLIPVisionModelWithProjection_fp16",
filename = "CLIP-ViT-H-fp16.safetensors",
local_dir = os.path.join(folder_paths.models_dir, "clip_vision"),
local_dir_use_symlinks=False)
else:
filename = "open_clip_pytorch_model.safetensors"
hf_hub_download(repo_id="laion/CLIP-ViT-H-14-laion2B-s32B-b79K",
filename = filename,
local_dir=download_path,
local_dir_use_symlinks=False)
source_file_path = os.path.join(download_path, filename)
destination_file_path = model_path
shutil.move(source_file_path, destination_file_path)
clip_vision = comfy.clip_vision.load(model_path)
print(f"Loading model from: {model_path}")
return (clip_vision,)
class DynamiCrafterModelLoader:
@classmethod
def INPUT_TYPES(s):
@@ -154,9 +233,6 @@ class DynamiCrafterModelLoader:
}),
"fp8_unet": ("BOOLEAN", {"default": False}),
},
"optional": {
"opt_openclippath": ("OPENCLIPVISIONPATH",)
}
}
RETURN_TYPES = ("DCMODEL",)
@@ -164,7 +240,7 @@ class DynamiCrafterModelLoader:
FUNCTION = "loadmodel"
CATEGORY = "DynamiCrafterWrapper"
def loadmodel(self, dtype, ckpt_name, fp8_unet=False, opt_openclippath=None):
def loadmodel(self, dtype, ckpt_name, fp8_unet=False):
mm.soft_empty_cache()
custom_config = {
'dtype': dtype,
@@ -190,11 +266,6 @@ class DynamiCrafterModelLoader:
print(f"No matching config for model: {ckpt_name}")
config = OmegaConf.load(config_file)
if opt_openclippath is not None:
print("Using open clip from: ", opt_openclippath)
config.model.params.cond_stage_config.params.version = opt_openclippath
config.model.params.img_cond_stage_config.params.version = opt_openclippath
model_config = config.pop("model", OmegaConf.create())
model_config['params']['unet_config']['params']['use_checkpoint']=False
self.model = instantiate_from_config(model_config)
@@ -222,12 +293,14 @@ class DynamiCrafterI2V:
def INPUT_TYPES(s):
return {"required": {
"model": ("DCMODEL",),
"clip_vision": ("CLIP_VISION", ),
"positive": ("CONDITIONING", ),
"negative": ("CONDITIONING", ),
"image": ("IMAGE",),
"steps": ("INT", {"default": 50, "min": 1, "max": 200, "step": 1}),
"cfg": ("FLOAT", {"default": 7.0, "min": 0.0, "max": 20.0, "step": 0.01}),
"eta": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
"frames": ("INT", {"default": 16, "min": 1, "max": 100, "step": 1}),
"prompt": ("STRING", {"multiline": True, "default": "",}),
"seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}),
"fs": ("INT", {"default": 10, "min": 2, "max": 100, "step": 1}),
"keep_model_loaded": ("BOOLEAN", {"default": True}),
@@ -255,8 +328,9 @@ class DynamiCrafterI2V:
FUNCTION = "process"
CATEGORY = "DynamiCrafterWrapper"
def process(self, model, image, prompt, cfg, steps, eta, seed, fs, keep_model_loaded, frames, vae_dtype, frame_window_size=16, frame_window_stride=4, mask=None, image2=None):
def process(self, model, image, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded, frames, vae_dtype, frame_window_size=16, frame_window_stride=4, mask=None, image2=None):
device = mm.get_torch_device()
offload_device = mm.unet_offload_device()
mm.unload_all_models()
mm.soft_empty_cache()
@@ -310,15 +384,15 @@ class DynamiCrafterI2V:
else:
img_tensor_repeat = repeat(z, 'b c t h w -> b c (repeat t) h w', repeat=frames)
self.model.first_stage_model.to('cpu')
self.model.cond_stage_model.to(device)
self.model.embedder.to(device)
self.model.first_stage_model.to(offload_device)
self.model.image_proj_model.to(device)
text_emb = positive[0][0].to(device)
cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
text_emb = self.model.get_learned_conditioning([prompt])
cond_images = self.model.embedder(image)
img_emb = self.model.image_proj_model(cond_images)
imtext_cond = torch.cat([text_emb, img_emb], dim=1)
del cond_images, img_emb, text_emb
@@ -334,12 +408,12 @@ class DynamiCrafterI2V:
## construct unconditional guidance
if cfg != 1.0:
uc_emb = self.model.get_learned_conditioning([""])
uc_emb = negative[0][0].to(device)
## process image embedding token
if hasattr(self.model, 'embedder'):
uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device)
## img: b c h w >> b l c
uc_img = self.model.embedder(uc_img)
uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(self.model.device)
uc_img = self.model.image_proj_model(uc_img)
uc_emb = torch.cat([uc_emb, uc_img], dim=1)
if isinstance(cond, dict):
@@ -350,9 +424,7 @@ class DynamiCrafterI2V:
else:
uc = None
self.model.cond_stage_model.to('cpu')
self.model.embedder.to('cpu')
self.model.image_proj_model.to('cpu')
self.model.image_proj_model.to(offload_device)
if mask is not None:
mask = mask.to(dtype).to(device)
@@ -395,8 +467,9 @@ class DynamiCrafterI2V:
## reconstruct from latent to pixel space
self.model.first_stage_model.to(device)
self.model.en_and_decode_n_samples_a_time = 1
decoded_images = self.model.decode_first_stage(samples) #b c t h w
self.model.first_stage_model.to('cpu')
self.model.first_stage_model.to(offload_device)
video = decoded_images.detach().cpu()
video = torch.clamp(video.float(), -1., 1.)
@@ -405,7 +478,7 @@ class DynamiCrafterI2V:
del decoded_images, samples
if not keep_model_loaded:
self.model.to('cpu')
self.model.to(offload_device)
mm.soft_empty_cache()
# Ensure the final dimensions are divisible by 2
final_H = (orig_H // 2) * 2
@@ -429,7 +502,6 @@ class ToonCrafterInterpolation:
"cfg": ("FLOAT", {"default": 7.0, "min": 0.0, "max": 200.0, "step": 0.01}),
"eta": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
"frames": ("INT", {"default": 16, "min": 1, "max": 100, "step": 1}),
"prompt": ("STRING", {"multiline": True, "default": "",}),
"seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}),
"fs": ("INT", {"default": 10, "min": 2, "max": 100, "step": 1}),
"vae_dtype": (
@@ -452,8 +524,9 @@ class ToonCrafterInterpolation:
FUNCTION = "process"
CATEGORY = "DynamiCrafterWrapper"
def process(self, model, clip_vision, images, positive, negative, prompt, cfg, steps, eta, seed, fs, frames, vae_dtype, image_embed_ratio=1.0):
def process(self, model, clip_vision, images, positive, negative, cfg, steps, eta, seed, fs, frames, vae_dtype, image_embed_ratio=1.0):
device = mm.get_torch_device()
offload_device = mm.unet_offload_device()
mm.unload_all_models()
mm.soft_empty_cache()
@@ -514,25 +587,17 @@ class ToonCrafterInterpolation:
img_tensor_repeat[:,:,:1,:,:] = z[:,:,:1,:,:]
img_tensor_repeat[:,:,-1:,:,:] = z[:,:,-1:,:,:]
self.model.first_stage_model.to('cpu')
#self.model.cond_stage_model.to(device)
#self.model.embedder.to(device)
#text_emb = self.model.get_learned_conditioning([prompt])
self.model.first_stage_model.to(offload_device)
text_emb = positive[0][0].to(device)
#cond_images = self.model.embedder(image)
#cond_images2 = self.model.embedder(image2)
cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
cond_images2 = clip_vision.encode_image(image2.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
self.model.image_proj_model.to(device)
img_emb = self.model.image_proj_model(cond_images)
img_emb2 = self.model.image_proj_model(cond_images2)
img_embeds = img_emb * image_embed_ratio + img_emb2 * (1.0 - image_embed_ratio)
imtext_cond = torch.cat([text_emb, img_embeds], dim=1)
@@ -550,15 +615,12 @@ class ToonCrafterInterpolation:
## construct unconditional guidance
if cfg != 1.0:
#uc_emb = self.model.get_learned_conditioning([""])
uc_emb = negative[0][0].to(device)
## process image embedding token
if hasattr(self.model, 'embedder'):
uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device)
## img: b c h w >> b l c
#uc_img = self.model.embedder(uc_img)
uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state']
uc_img = uc_img.to(self.model.device)
uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(self.model.device)
uc_img = self.model.image_proj_model(uc_img)
uc_emb = torch.cat([uc_emb, uc_img], dim=1)
if isinstance(cond, dict):
@@ -569,9 +631,7 @@ class ToonCrafterInterpolation:
else:
uc = None
#self.model.cond_stage_model.to('cpu')
#self.model.embedder.to('cpu')
self.model.image_proj_model.to('cpu')
self.model.image_proj_model.to(offload_device)
#inference
@@ -602,7 +662,7 @@ class ToonCrafterInterpolation:
samples = samples.squeeze(0).permute(1, 0, 2, 3)
out.append(samples)
self.model.to('cpu')
self.model.to(offload_device)
mm.soft_empty_cache()
samples = torch.cat(out, dim=0)
@@ -648,6 +708,8 @@ class ToonCrafterDecode:
samples = samples * 0.18215
hs = latent["hidden_states"]
model.en_and_decode_n_samples_a_time = 16
if vae_dtype == "auto":
try:
if mm.should_use_bf16():
@@ -901,16 +963,18 @@ NODE_CLASS_MAPPINGS = {
"DynamiCrafterBatchInterpolation": DynamiCrafterBatchInterpolation,
"ToonCrafterInterpolation": ToonCrafterInterpolation,
"ToonCrafterDecode": ToonCrafterDecode,
"OpenCLIPVisionSelect": OpenCLIPVisionSelect,
"DownloadAndLoadDynamiCrafterModel": DownloadAndLoadDynamiCrafterModel
"DownloadAndLoadDynamiCrafterModel": DownloadAndLoadDynamiCrafterModel,
"DownloadAndLoadCLIPModel": DownloadAndLoadCLIPModel,
"DownloadAndLoadCLIPVisionModel": DownloadAndLoadCLIPVisionModel
}
NODE_DISPLAY_NAME_MAPPINGS = {
"DynamiCrafterI2V": "DynamiCrafterI2V",
"DynamiCrafterModelLoader": "DynamiCrafterModelLoader",
"DynamiCrafterBatchInterpolation": "DynamiCrafterBatchInterpolation",
"OpenCLIPVisionSelect": "OpenCLIPVisionSelect",
"ToonCrafterInterpolation": "ToonCrafterInterpolation",
"ToonCrafterDecode": "ToonCrafterDecode",
"DownloadAndLoadDynamiCrafterModel": "DownloadAndLoadDynamiCrafterModel"
"DownloadAndLoadDynamiCrafterModel": "DownloadAndLoadDynamiCrafterModel",
"DownloadAndLoadCLIPModel": "DownloadAndLoadCLIPModel",
"DownloadAndLoadCLIPVisionModel": "DownloadAndLoadCLIPVisionModel"
}