Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8eddf6ed3d | ||
|
|
c17a198800 | ||
|
|
feb9d28f1d | ||
|
|
c5abb6f538 |
@@ -45,6 +45,8 @@ conda install cuda-nvcc -c nvidia
|
|||||||
```
|
```
|
||||||
|
|
||||||

|

|
||||||
|

|
||||||
|

|
||||||

|

|
||||||
|
|
||||||
根据自己系统选择 Windows 10 SDK / Windows 11 SDK.
|
根据自己系统选择 Windows 10 SDK / Windows 11 SDK.
|
||||||
@@ -64,38 +66,15 @@ vcvars64.bat
|
|||||||
|
|
||||||
编译完成,成功启动。
|
编译完成,成功启动。
|
||||||
|
|
||||||
## diffusers 版本
|
|
||||||
|
|
||||||
`main` 分支锁定 diffusers==0.24
|
|
||||||
|
|
||||||
`diffusers-0.26` 分支锁定 diffusers==0.26.x
|
|
||||||
|
|
||||||
要切换分支,请使用下面命令:
|
|
||||||
|
|
||||||
```
|
|
||||||
git switch diffusers-0.26
|
|
||||||
```
|
|
||||||
|
|
||||||
并重新安装依赖:
|
|
||||||
|
|
||||||
```
|
|
||||||
pip install --force-reinstall -r custom_nodes/ComfyUI-OOTDiffusion/requirements.txt
|
|
||||||
```
|
|
||||||
|
|
||||||
## FAQ 常见错误
|
## FAQ 常见错误
|
||||||
|
|
||||||
```
|
> fatal error: cuda_runtime.h: No such file or directory compilation terminated. ninja: build stopped: subcommand failed.
|
||||||
fatal error: cuda_runtime.h: No such file or directory compilation terminated.
|
>
|
||||||
ninja: build stopped: subcommand failed.
|
> 解决办法:`conda install cuda-toolkit=12.1 -c nvidia`
|
||||||
```
|
|
||||||
|
|
||||||
解决办法:`conda install cuda-toolkit=12.1 -c nvidia` 并覆写 `CUDA_HOME` `CUDA_PATH` 环境变量
|
> subprocess.CalledProcessError: Command '['where', 'cl']' returned non-zero exit status 1.
|
||||||
|
>
|
||||||
```
|
> 解决办法:仅在 Windows 下出现,根据 [Windows 配置教程](#Windows-指南)
|
||||||
subprocess.CalledProcessError: Command '['where', 'cl']' returned non-zero exit status 1.
|
|
||||||
```
|
|
||||||
|
|
||||||
解决办法:仅在 Windows 下出现,根据 [Windows 配置教程](#Windows-指南)
|
|
||||||
|
|
||||||
## Node 节点
|
## Node 节点
|
||||||
|
|
||||||
@@ -105,8 +84,6 @@ Load OOTDiffusion from Hub: 从 huggingface 自动下载并加载 OOTDiffusion P
|
|||||||
|
|
||||||
OOTDiffusion Generate: 生成图像
|
OOTDiffusion Generate: 生成图像
|
||||||
|
|
||||||
参数:
|
|
||||||
|
|
||||||
cfg: 输出图像和输入衣服的贴合程度
|
cfg: 输出图像和输入衣服的贴合程度
|
||||||
|
|
||||||
## Example image 示例图片
|
## Example image 示例图片
|
||||||
@@ -117,16 +94,16 @@ Full body 全身: [模特](./assets/model_fullbody_1.png) [裤子](./assets/clot
|
|||||||
|
|
||||||
Full body 裙子: [模特](./assets/model_dress_1.png) [裙子](./assets/cloth_dress_1.jpg)
|
Full body 裙子: [模特](./assets/model_dress_1.png) [裙子](./assets/cloth_dress_1.jpg)
|
||||||
|
|
||||||
|
## Detail 细节
|
||||||
|
|
||||||
|
目前此项目只是对 OOTDiffusion 的功能做了个简单的迁移。
|
||||||
|
OOTDiffusion 本体依赖于 `diffusers==0.24.0` 实现,所以假如有其他节点的依赖冲突是没办法解决的(本就不该依赖 diffusers)。
|
||||||
|
靠 vendor 也能解决,所以也不是大问题。
|
||||||
|
|
||||||
|
在 `Ubuntu 22.02` / `Python 3.10.x` 下可以正常运行。Windows 没有测试过。
|
||||||
|
|
||||||
## 更新日志 Release Note
|
## 更新日志 Release Note
|
||||||
|
|
||||||
2024-03-14:
|
|
||||||
|
|
||||||
添加 `diffusers-0.26` 分支
|
|
||||||
|
|
||||||
2024-03-10:
|
|
||||||
|
|
||||||
添加 humanparsing onnx 支持
|
|
||||||
|
|
||||||
2024-03-04:
|
2024-03-04:
|
||||||
|
|
||||||
添加 Full body 模型
|
添加 Full body 模型
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ from diffusers.configuration_utils import ConfigMixin, register_to_config
|
|||||||
from diffusers.models.embeddings import ImagePositionalEmbeddings
|
from diffusers.models.embeddings import ImagePositionalEmbeddings
|
||||||
from diffusers.utils import USE_PEFT_BACKEND, BaseOutput, deprecate
|
from diffusers.utils import USE_PEFT_BACKEND, BaseOutput, deprecate
|
||||||
# from diffusers.models.attention import BasicTransformerBlock
|
# from diffusers.models.attention import BasicTransformerBlock
|
||||||
from diffusers.models.embeddings import CaptionProjection, PatchEmbed
|
from diffusers.models.embeddings import PixArtAlphaTextProjection, PatchEmbed
|
||||||
from diffusers.models.lora import LoRACompatibleConv, LoRACompatibleLinear
|
from diffusers.models.lora import LoRACompatibleConv, LoRACompatibleLinear
|
||||||
from diffusers.models.modeling_utils import ModelMixin
|
from diffusers.models.modeling_utils import ModelMixin
|
||||||
from diffusers.models.normalization import AdaLayerNormSingle
|
from diffusers.models.normalization import AdaLayerNormSingle
|
||||||
@@ -237,7 +237,7 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
|
|||||||
|
|
||||||
self.caption_projection = None
|
self.caption_projection = None
|
||||||
if caption_channels is not None:
|
if caption_channels is not None:
|
||||||
self.caption_projection = CaptionProjection(in_features=caption_channels, hidden_size=inner_dim)
|
self.caption_projection = PixArtAlphaTextProjection(in_features=caption_channels, hidden_size=inner_dim)
|
||||||
|
|
||||||
self.gradient_checkpointing = False
|
self.gradient_checkpointing = False
|
||||||
|
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ from diffusers.configuration_utils import ConfigMixin, register_to_config
|
|||||||
from diffusers.models.embeddings import ImagePositionalEmbeddings
|
from diffusers.models.embeddings import ImagePositionalEmbeddings
|
||||||
from diffusers.utils import USE_PEFT_BACKEND, BaseOutput, deprecate
|
from diffusers.utils import USE_PEFT_BACKEND, BaseOutput, deprecate
|
||||||
# from diffusers.models.attention import BasicTransformerBlock
|
# from diffusers.models.attention import BasicTransformerBlock
|
||||||
from diffusers.models.embeddings import CaptionProjection, PatchEmbed
|
from diffusers.models.embeddings import PixArtAlphaTextProjection, PatchEmbed
|
||||||
from diffusers.models.lora import LoRACompatibleConv, LoRACompatibleLinear
|
from diffusers.models.lora import LoRACompatibleConv, LoRACompatibleLinear
|
||||||
from diffusers.models.modeling_utils import ModelMixin
|
from diffusers.models.modeling_utils import ModelMixin
|
||||||
from diffusers.models.normalization import AdaLayerNormSingle
|
from diffusers.models.normalization import AdaLayerNormSingle
|
||||||
@@ -237,7 +237,7 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
|
|||||||
|
|
||||||
self.caption_projection = None
|
self.caption_projection = None
|
||||||
if caption_channels is not None:
|
if caption_channels is not None:
|
||||||
self.caption_projection = CaptionProjection(in_features=caption_channels, hidden_size=inner_dim)
|
self.caption_projection = PixArtAlphaTextProjection(in_features=caption_channels, hidden_size=inner_dim)
|
||||||
|
|
||||||
self.gradient_checkpointing = False
|
self.gradient_checkpointing = False
|
||||||
|
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ from diffusers.models.embeddings import (
|
|||||||
ImageHintTimeEmbedding,
|
ImageHintTimeEmbedding,
|
||||||
ImageProjection,
|
ImageProjection,
|
||||||
ImageTimeEmbedding,
|
ImageTimeEmbedding,
|
||||||
PositionNet,
|
GLIGENTextBoundingboxProjection,
|
||||||
TextImageProjection,
|
TextImageProjection,
|
||||||
TextImageTimeEmbedding,
|
TextImageTimeEmbedding,
|
||||||
TextTimeEmbedding,
|
TextTimeEmbedding,
|
||||||
@@ -624,7 +624,7 @@ class UNetGarm2DConditionModel(ModelMixin, ConfigMixin, UNet2DConditionLoadersMi
|
|||||||
positive_len = cross_attention_dim[0]
|
positive_len = cross_attention_dim[0]
|
||||||
|
|
||||||
feature_type = "text-only" if attention_type == "gated" else "text-image"
|
feature_type = "text-only" if attention_type == "gated" else "text-image"
|
||||||
self.position_net = PositionNet(
|
self.position_net = GLIGENTextBoundingboxProjection(
|
||||||
positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type
|
positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ from diffusers.models.embeddings import (
|
|||||||
ImageHintTimeEmbedding,
|
ImageHintTimeEmbedding,
|
||||||
ImageProjection,
|
ImageProjection,
|
||||||
ImageTimeEmbedding,
|
ImageTimeEmbedding,
|
||||||
PositionNet,
|
GLIGENTextBoundingboxProjection,
|
||||||
TextImageProjection,
|
TextImageProjection,
|
||||||
TextImageTimeEmbedding,
|
TextImageTimeEmbedding,
|
||||||
TextTimeEmbedding,
|
TextTimeEmbedding,
|
||||||
@@ -624,7 +624,7 @@ class UNetVton2DConditionModel(ModelMixin, ConfigMixin, UNet2DConditionLoadersMi
|
|||||||
positive_len = cross_attention_dim[0]
|
positive_len = cross_attention_dim[0]
|
||||||
|
|
||||||
feature_type = "text-only" if attention_type == "gated" else "text-image"
|
feature_type = "text-only" if attention_type == "gated" else "text-image"
|
||||||
self.position_net = PositionNet(
|
self.position_net = GLIGENTextBoundingboxProjection(
|
||||||
positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type
|
positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ scipy
|
|||||||
scikit-image
|
scikit-image
|
||||||
opencv-python
|
opencv-python
|
||||||
pillow
|
pillow
|
||||||
diffusers==0.24.0
|
diffusers>=0.26.0
|
||||||
transformers
|
transformers
|
||||||
accelerate
|
accelerate
|
||||||
matplotlib
|
matplotlib
|
||||||
|
|||||||
Reference in New Issue
Block a user