diff --git a/README.md b/README.md index bd320a3..5a950bf 100644 --- a/README.md +++ b/README.md @@ -45,10 +45,10 @@ Available models: | From | to `v1` | to `xl` | to `v3` | to `ca` | |:----:|:-------:|:-------:|:-------:|:-------:| -| `v1` | - | v4.0 | No | No | -| `xl` | v4.0 | - | No | No | +| `v1` | - | v4.0 | v4.0 | No | +| `xl` | v4.0 | - | v4.0 | No | | `v3` | v4.0 | v4.0 | - | No | -| `ca` | v4.0 | v4.0 | No | - | +| `ca` | v4.0 | v4.0 | v4.0 | - | ## Training diff --git a/comfy_latent_interposer.py b/comfy_latent_interposer.py index 6c0d74a..a5a15cf 100644 --- a/comfy_latent_interposer.py +++ b/comfy_latent_interposer.py @@ -6,15 +6,19 @@ from huggingface_hub import hf_hub_download # v1 = Stable Diffusion 1.x # xl = Stable Diffusion Extra Large (SDXL) +# v3 = Stable Diffusion Version Three (SD3) # cc = Stable Cascade (Stage C) [not used] # ca = Stable Cascade (Stage A/B) config = { "v1-to-xl": {"ch_in": 4, "ch_out": 4, "ch_mid": 64, "scale": 1.0, "blocks": 12}, + "v1-to-v3": {"ch_in": 4, "ch_out":16, "ch_mid": 64, "scale": 1.0, "blocks": 12}, "xl-to-v1": {"ch_in": 4, "ch_out": 4, "ch_mid": 64, "scale": 1.0, "blocks": 12}, - "ca-to-v1": {"ch_in": 4, "ch_out": 4, "ch_mid": 64, "scale": 0.5, "blocks": 12}, - "ca-to-xl": {"ch_in": 4, "ch_out": 4, "ch_mid": 64, "scale": 0.5, "blocks": 12}, + "xl-to-v3": {"ch_in": 4, "ch_out":16, "ch_mid": 64, "scale": 1.0, "blocks": 12}, "v3-to-v1": {"ch_in":16, "ch_out": 4, "ch_mid": 64, "scale": 1.0, "blocks": 12}, "v3-to-xl": {"ch_in":16, "ch_out": 4, "ch_mid": 64, "scale": 1.0, "blocks": 12}, + "ca-to-v1": {"ch_in": 4, "ch_out": 4, "ch_mid": 64, "scale": 0.5, "blocks": 12}, + "ca-to-xl": {"ch_in": 4, "ch_out": 4, "ch_mid": 64, "scale": 0.5, "blocks": 12}, + "ca-to-v3": {"ch_in": 4, "ch_out":16, "ch_mid": 64, "scale": 0.5, "blocks": 12}, } class ResBlock(nn.Module): @@ -92,7 +96,7 @@ class ComfyLatentInterposer: "required": { "samples": ("LATENT", ), "latent_src": (["v1", "xl", "v3", "ca"],), - "latent_dst": (["v1", "xl"],), + "latent_dst": (["v1", "xl", "v3"],), } } diff --git a/config/ca-to-v3.yaml b/config/ca-to-v3.yaml new file mode 100644 index 0000000..f273ede --- /dev/null +++ b/config/ca-to-v3.yaml @@ -0,0 +1,40 @@ +steps: 20000 +batch: 48 +fconst: 0 +cosine: False +resume: False +device: "cuda" +p_loss_weight: 1.0 +r_loss_weight: 0.1 +b_loss_weight: 1.0 +h_loss_weight: 0.0 +save_image: 100 +eval_model: 10 + +model: + src: ca # Stable Cascade Stage A + dst: v3 # Stable Diffusion Version three point oh + rev: "v4.0-rc1" + args: + scale: 0.5 + ch_in: 4 + ch_out: 16 + ch_mid: 64 + blocks: 12 + +optim: + lr: 5.0e-4 + beta1: 0.5 + beta2: 0.95 + +dataset: + src: "./latents/ca_256px_combined.bin" + dst: "./latents/v3_256px_combined.bin" + preload: False + evals: + main: + src: "./latents/test_eru/test_ca_768px.npy" + dst: "./latents/test_eru/test_v3_768px.npy" + aux: + src: "./latents/test_bga/test_ca_768px.npy" + dst: "./latents/test_bga/test_v3_768px.npy" diff --git a/config/v1-to-v3.yaml b/config/v1-to-v3.yaml new file mode 100644 index 0000000..6663943 --- /dev/null +++ b/config/v1-to-v3.yaml @@ -0,0 +1,40 @@ +steps: 20000 +batch: 128 +fconst: 0 +cosine: False +resume: False +device: "cuda" +p_loss_weight: 1.0 +r_loss_weight: 0.1 +b_loss_weight: 1.0 +h_loss_weight: 0.0 +save_image: 100 +eval_model: 10 + +model: + src: v1 # Stable Diffusion 1.x + dst: v3 # Stable Diffusion Version three point oh + rev: "v4.0-rc5" + args: + scale: 1.0 + ch_in: 4 + ch_out: 16 + ch_mid: 64 + blocks: 12 + +optim: + lr: 5.0e-4 + beta1: 0.5 + beta2: 0.95 + +dataset: + src: "./latents/v1_256px_combined.bin" + dst: "./latents/v3_256px_combined.bin" + preload: False + evals: + main: + src: "./latents/test_eru/test_v1_768px.npy" + dst: "./latents/test_eru/test_v3_768px.npy" + aux: + src: "./latents/test_bga/test_v1_768px.npy" + dst: "./latents/test_bga/test_v3_768px.npy" diff --git a/config/xl-to-v3.yaml b/config/xl-to-v3.yaml new file mode 100644 index 0000000..9dcafe8 --- /dev/null +++ b/config/xl-to-v3.yaml @@ -0,0 +1,40 @@ +steps: 20000 +batch: 128 +fconst: 0 +cosine: False +resume: False +device: "cuda" +p_loss_weight: 1.0 +r_loss_weight: 0.1 +b_loss_weight: 1.0 +h_loss_weight: 0.0 +save_image: 100 +eval_model: 10 + +model: + src: xl # Stable Diffusion Extra Large + dst: v3 # Stable Diffusion Version three point oh + rev: "v4.0-rc5" + args: + scale: 1.0 + ch_in: 4 + ch_out: 16 + ch_mid: 64 + blocks: 12 + +optim: + lr: 5.0e-4 + beta1: 0.5 + beta2: 0.95 + +dataset: + src: "./latents/xl_256px_combined.bin" + dst: "./latents/v3_256px_combined.bin" + preload: False + evals: + main: + src: "./latents/test_eru/test_xl_768px.npy" + dst: "./latents/test_eru/test_v3_768px.npy" + aux: + src: "./latents/test_bga/test_xl_768px.npy" + dst: "./latents/test_bga/test_v3_768px.npy"