Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8eddf6ed3d | ||
|
|
c17a198800 | ||
|
|
feb9d28f1d | ||
|
|
c5abb6f538 |
@@ -45,6 +45,8 @@ conda install cuda-nvcc -c nvidia
|
||||
```
|
||||
|
||||

|
||||

|
||||

|
||||

|
||||
|
||||
根据自己系统选择 Windows 10 SDK / Windows 11 SDK.
|
||||
@@ -64,38 +66,15 @@ vcvars64.bat
|
||||
|
||||
编译完成,成功启动。
|
||||
|
||||
## diffusers 版本
|
||||
|
||||
`main` 分支锁定 diffusers==0.24
|
||||
|
||||
`diffusers-0.26` 分支锁定 diffusers==0.26.x
|
||||
|
||||
要切换分支,请使用下面命令:
|
||||
|
||||
```
|
||||
git switch diffusers-0.26
|
||||
```
|
||||
|
||||
并重新安装依赖:
|
||||
|
||||
```
|
||||
pip install --force-reinstall -r custom_nodes/ComfyUI-OOTDiffusion/requirements.txt
|
||||
```
|
||||
|
||||
## FAQ 常见错误
|
||||
|
||||
```
|
||||
fatal error: cuda_runtime.h: No such file or directory compilation terminated.
|
||||
ninja: build stopped: subcommand failed.
|
||||
```
|
||||
> fatal error: cuda_runtime.h: No such file or directory compilation terminated. ninja: build stopped: subcommand failed.
|
||||
>
|
||||
> 解决办法:`conda install cuda-toolkit=12.1 -c nvidia`
|
||||
|
||||
解决办法:`conda install cuda-toolkit=12.1 -c nvidia` 并覆写 `CUDA_HOME` `CUDA_PATH` 环境变量
|
||||
|
||||
```
|
||||
subprocess.CalledProcessError: Command '['where', 'cl']' returned non-zero exit status 1.
|
||||
```
|
||||
|
||||
解决办法:仅在 Windows 下出现,根据 [Windows 配置教程](#Windows-指南)
|
||||
> subprocess.CalledProcessError: Command '['where', 'cl']' returned non-zero exit status 1.
|
||||
>
|
||||
> 解决办法:仅在 Windows 下出现,根据 [Windows 配置教程](#Windows-指南)
|
||||
|
||||
## Node 节点
|
||||
|
||||
@@ -105,8 +84,6 @@ Load OOTDiffusion from Hub: 从 huggingface 自动下载并加载 OOTDiffusion P
|
||||
|
||||
OOTDiffusion Generate: 生成图像
|
||||
|
||||
参数:
|
||||
|
||||
cfg: 输出图像和输入衣服的贴合程度
|
||||
|
||||
## Example image 示例图片
|
||||
@@ -117,16 +94,16 @@ Full body 全身: [模特](./assets/model_fullbody_1.png) [裤子](./assets/clot
|
||||
|
||||
Full body 裙子: [模特](./assets/model_dress_1.png) [裙子](./assets/cloth_dress_1.jpg)
|
||||
|
||||
## Detail 细节
|
||||
|
||||
目前此项目只是对 OOTDiffusion 的功能做了个简单的迁移。
|
||||
OOTDiffusion 本体依赖于 `diffusers==0.24.0` 实现,所以假如有其他节点的依赖冲突是没办法解决的(本就不该依赖 diffusers)。
|
||||
靠 vendor 也能解决,所以也不是大问题。
|
||||
|
||||
在 `Ubuntu 22.02` / `Python 3.10.x` 下可以正常运行。Windows 没有测试过。
|
||||
|
||||
## 更新日志 Release Note
|
||||
|
||||
2024-03-14:
|
||||
|
||||
添加 `diffusers-0.26` 分支
|
||||
|
||||
2024-03-10:
|
||||
|
||||
添加 humanparsing onnx 支持
|
||||
|
||||
2024-03-04:
|
||||
|
||||
添加 Full body 模型
|
||||
|
||||
@@ -26,7 +26,7 @@ from diffusers.configuration_utils import ConfigMixin, register_to_config
|
||||
from diffusers.models.embeddings import ImagePositionalEmbeddings
|
||||
from diffusers.utils import USE_PEFT_BACKEND, BaseOutput, deprecate
|
||||
# from diffusers.models.attention import BasicTransformerBlock
|
||||
from diffusers.models.embeddings import CaptionProjection, PatchEmbed
|
||||
from diffusers.models.embeddings import PixArtAlphaTextProjection, PatchEmbed
|
||||
from diffusers.models.lora import LoRACompatibleConv, LoRACompatibleLinear
|
||||
from diffusers.models.modeling_utils import ModelMixin
|
||||
from diffusers.models.normalization import AdaLayerNormSingle
|
||||
@@ -237,7 +237,7 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
|
||||
|
||||
self.caption_projection = None
|
||||
if caption_channels is not None:
|
||||
self.caption_projection = CaptionProjection(in_features=caption_channels, hidden_size=inner_dim)
|
||||
self.caption_projection = PixArtAlphaTextProjection(in_features=caption_channels, hidden_size=inner_dim)
|
||||
|
||||
self.gradient_checkpointing = False
|
||||
|
||||
|
||||
@@ -26,7 +26,7 @@ from diffusers.configuration_utils import ConfigMixin, register_to_config
|
||||
from diffusers.models.embeddings import ImagePositionalEmbeddings
|
||||
from diffusers.utils import USE_PEFT_BACKEND, BaseOutput, deprecate
|
||||
# from diffusers.models.attention import BasicTransformerBlock
|
||||
from diffusers.models.embeddings import CaptionProjection, PatchEmbed
|
||||
from diffusers.models.embeddings import PixArtAlphaTextProjection, PatchEmbed
|
||||
from diffusers.models.lora import LoRACompatibleConv, LoRACompatibleLinear
|
||||
from diffusers.models.modeling_utils import ModelMixin
|
||||
from diffusers.models.normalization import AdaLayerNormSingle
|
||||
@@ -237,7 +237,7 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
|
||||
|
||||
self.caption_projection = None
|
||||
if caption_channels is not None:
|
||||
self.caption_projection = CaptionProjection(in_features=caption_channels, hidden_size=inner_dim)
|
||||
self.caption_projection = PixArtAlphaTextProjection(in_features=caption_channels, hidden_size=inner_dim)
|
||||
|
||||
self.gradient_checkpointing = False
|
||||
|
||||
|
||||
@@ -44,7 +44,7 @@ from diffusers.models.embeddings import (
|
||||
ImageHintTimeEmbedding,
|
||||
ImageProjection,
|
||||
ImageTimeEmbedding,
|
||||
PositionNet,
|
||||
GLIGENTextBoundingboxProjection,
|
||||
TextImageProjection,
|
||||
TextImageTimeEmbedding,
|
||||
TextTimeEmbedding,
|
||||
@@ -624,7 +624,7 @@ class UNetGarm2DConditionModel(ModelMixin, ConfigMixin, UNet2DConditionLoadersMi
|
||||
positive_len = cross_attention_dim[0]
|
||||
|
||||
feature_type = "text-only" if attention_type == "gated" else "text-image"
|
||||
self.position_net = PositionNet(
|
||||
self.position_net = GLIGENTextBoundingboxProjection(
|
||||
positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type
|
||||
)
|
||||
|
||||
|
||||
@@ -44,7 +44,7 @@ from diffusers.models.embeddings import (
|
||||
ImageHintTimeEmbedding,
|
||||
ImageProjection,
|
||||
ImageTimeEmbedding,
|
||||
PositionNet,
|
||||
GLIGENTextBoundingboxProjection,
|
||||
TextImageProjection,
|
||||
TextImageTimeEmbedding,
|
||||
TextTimeEmbedding,
|
||||
@@ -624,7 +624,7 @@ class UNetVton2DConditionModel(ModelMixin, ConfigMixin, UNet2DConditionLoadersMi
|
||||
positive_len = cross_attention_dim[0]
|
||||
|
||||
feature_type = "text-only" if attention_type == "gated" else "text-image"
|
||||
self.position_net = PositionNet(
|
||||
self.position_net = GLIGENTextBoundingboxProjection(
|
||||
positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type
|
||||
)
|
||||
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@ scipy
|
||||
scikit-image
|
||||
opencv-python
|
||||
pillow
|
||||
diffusers==0.24.0
|
||||
diffusers>=0.26.0
|
||||
transformers
|
||||
accelerate
|
||||
matplotlib
|
||||
|
||||
Reference in New Issue
Block a user