Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b242449b25 | ||
|
|
2fb8fd1872 | ||
|
|
ef82a32944 | ||
|
|
aa7e959330 | ||
|
|
edff6352d5 | ||
|
|
aa8250e5d2 | ||
|
|
e7255aac25 | ||
|
|
b752ff1cbc | ||
|
|
8b08629bd0 | ||
|
|
c70ef0fc47 | ||
|
|
8076aae7da | ||
|
|
79e4910ce4 | ||
|
|
6be496d5ac | ||
|
|
b5b2fd9974 | ||
|
|
f502f8063d | ||
|
|
a3711e1d1d | ||
|
|
8ba0b4c673 | ||
|
|
4f4e516164 | ||
|
|
88f0244516 | ||
|
|
5733918aec | ||
|
|
da88802b68 | ||
|
|
16d99a268f | ||
|
|
fc46a63ac8 | ||
|
|
0010d4282a | ||
|
|
36b259b4b4 | ||
|
|
2a7e026f84 |
@@ -15,6 +15,6 @@ dev
|
||||
scepter.egg-info
|
||||
.readthedocs.yml
|
||||
1.9
|
||||
MANIFEST.in
|
||||
#MANIFEST.in
|
||||
*resources
|
||||
*.ipynb_checkpoints*
|
||||
|
||||
@@ -1 +1,2 @@
|
||||
recursive-include scepter *.yaml
|
||||
recursive-include scepter *.md
|
||||
|
||||
|
After Width: | Height: | Size: 90 KiB |
|
After Width: | Height: | Size: 72 KiB |
|
After Width: | Height: | Size: 62 KiB |
|
After Width: | Height: | Size: 66 KiB |
|
After Width: | Height: | Size: 66 KiB |
|
After Width: | Height: | Size: 27 KiB |
|
After Width: | Height: | Size: 87 KiB |
|
After Width: | Height: | Size: 91 KiB |
|
After Width: | Height: | Size: 27 KiB |
|
After Width: | Height: | Size: 72 KiB |
|
After Width: | Height: | Size: 64 KiB |
|
Before Width: | Height: | Size: 118 KiB After Width: | Height: | Size: 109 KiB |
|
After Width: | Height: | Size: 247 KiB |
|
After Width: | Height: | Size: 301 KiB |
|
After Width: | Height: | Size: 230 KiB |
|
After Width: | Height: | Size: 353 KiB |
|
After Width: | Height: | Size: 32 KiB |
|
After Width: | Height: | Size: 144 KiB |
|
After Width: | Height: | Size: 126 KiB |
|
After Width: | Height: | Size: 138 KiB |
|
After Width: | Height: | Size: 170 KiB |
|
After Width: | Height: | Size: 273 KiB |
|
After Width: | Height: | Size: 243 KiB |
|
After Width: | Height: | Size: 134 KiB |
|
After Width: | Height: | Size: 119 KiB |
|
After Width: | Height: | Size: 433 KiB |
|
After Width: | Height: | Size: 265 KiB |
|
After Width: | Height: | Size: 300 KiB |
|
After Width: | Height: | Size: 155 KiB |
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
# Configuration file for the Sphinx documentation builder.
|
||||
#
|
||||
# This file only contains a selection of the most common options. For a full
|
||||
|
||||
@@ -14,8 +14,8 @@ Model modules are divided into backbones, necks, heads, loss, metrics, networks,
|
||||
Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import BACKBONES
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@BACKBONES.register_class("ResNet")
|
||||
@@ -25,8 +25,8 @@ class ResNet(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import NECKS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import NECKS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@NECKS.register_class()
|
||||
@@ -36,8 +36,8 @@ class GlobalAveragePooling(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import HEADS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import HEADS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@HEADS.register_class()
|
||||
@@ -47,7 +47,7 @@ class ClassifierHead(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import LOSSES
|
||||
from scepter.modules.model.registry import LOSSES
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
@@ -59,7 +59,7 @@ class CrossEntropy(nn.Module):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES, NECKS, HEADS, LOSSES
|
||||
from scepter.modules.model.registry import BACKBONES, NECKS, HEADS, LOSSES
|
||||
|
||||
backbone = BACKBONES.build(cfg.BACKBONE, logger=logger)
|
||||
neck = NECKS.build(cfg.NECK, logger=logger)
|
||||
@@ -83,8 +83,8 @@ To be implemented specifically as needed;
|
||||
Basic Usage Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.model.metrics.base_metric import BaseMetric
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.base_metric import BaseMetric
|
||||
|
||||
|
||||
@METRICS.register_class("AccuracyMetric")
|
||||
@@ -95,7 +95,7 @@ class AccuracyMetric(BaseMetric):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
|
||||
metric = METRICS.build(cfgs, logger)
|
||||
```
|
||||
@@ -117,8 +117,8 @@ Typically takes logits and labels as well as other necessary variables as inputs
|
||||
Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.model.tokenizers import BaseTokenizer
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.tokenizers import BaseTokenizer
|
||||
|
||||
|
||||
@TOKENIZERS.register_class()
|
||||
@@ -129,7 +129,7 @@ class BaseBertTokenizer(BaseTokenizer):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
|
||||
tokenizer = TOKENIZERS.build(cfgs, logger)
|
||||
```
|
||||
@@ -147,8 +147,8 @@ Takes a list of texts that need tokenization as input and outputs token id seque
|
||||
Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.model.networks.train_module import TrainModule
|
||||
from scepter.modules.model.registry import MODELS
|
||||
from scepter.modules.model.networks.train_module import TrainModule
|
||||
|
||||
|
||||
@MODELS.register_class()
|
||||
@@ -159,7 +159,7 @@ class Classifier(TrainModule):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.modules.model.registry import MODELS
|
||||
|
||||
model = MODELS.build(self.cfg.MODEL, logger=self.logger)
|
||||
```
|
||||
|
||||
@@ -9,8 +9,8 @@
|
||||
Usage when subclassing lr_schedulers:
|
||||
|
||||
```python
|
||||
from scepter.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
from scepter.modules.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.modules.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
|
||||
|
||||
@LR_SCHEDULERS.register_class()
|
||||
@@ -48,8 +48,8 @@ Sets up the schedule for the passed-in optimizer object;
|
||||
Usage when subclassing optimizers:
|
||||
|
||||
```python
|
||||
from scepter.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.opt.optimizers.registry import OPTIMIZERS
|
||||
from scepter.modules.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.modules.opt.optimizers.registry import OPTIMIZERS
|
||||
|
||||
|
||||
@OPTIMIZERS.register_class()
|
||||
|
||||
@@ -6,17 +6,17 @@ This is the File System Module, designed to handle file transfer functionalities
|
||||
|
||||
The component currently supports three types of IO Handler:
|
||||
|
||||
1. scepter.utils.file_clients.AliyunOssFs
|
||||
2. scepter.utils.file_clients.LocalFs
|
||||
3. scepter.utils.file_clients.HttpFs
|
||||
1. scepter.modules.utils.file_clients.AliyunOssFs
|
||||
2. scepter.modules.utils.file_clients.LocalFs
|
||||
3. scepter.modules.utils.file_clients.HttpFs
|
||||
|
||||
<hr/>
|
||||
|
||||
## Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
fs_cfg = Config(load=False, cfg_dict={
|
||||
"NAME": "AliyunOssFs",
|
||||
|
||||
@@ -4,18 +4,18 @@ Relies on SDKs, which are used to organize modules and SDKs that are frequently
|
||||
|
||||
## Overview
|
||||
|
||||
1. Parameter sdk (scepter.utils.config)
|
||||
2. Path sdk (scepter.utils.directory)
|
||||
3. PyTorch distributed sdk (scepter.utils.distribute)
|
||||
4. Model export sdk (scepter.utils.export_model)
|
||||
5. File system sdk (scepter.utils.file_system)
|
||||
6. Logging sdk (scepter.utils.logger)
|
||||
7. Video processing sdk (scepter.utils.video_reader), see the document (video_reader.md)
|
||||
8. Module registration sdk (scepter.utils.registry)
|
||||
9. Data sdk (scepter.utils.data)
|
||||
10. Model sdk (scepter.utils.model)
|
||||
11. Sampler sdk (scepter.utils.sampler)
|
||||
12. Probing sdk (scepter.utils.probe)
|
||||
1. Parameter sdk (scepter.modules.utils.config)
|
||||
2. Path sdk (scepter.modules.utils.directory)
|
||||
3. PyTorch distributed sdk (scepter.modules.utils.distribute)
|
||||
4. Model export sdk (scepter.modules.utils.export_model)
|
||||
5. File system sdk (scepter.modules.utils.file_system)
|
||||
6. Logging sdk (scepter.modules.utils.logger)
|
||||
7. Video processing sdk (scepter.modules.utils.video_reader), see the document (video_reader.md)
|
||||
8. Module registration sdk (scepter.modules.utils.registry)
|
||||
9. Data sdk (scepter.modules.utils.data)
|
||||
10. Model sdk (scepter.modules.utils.model)
|
||||
11. Sampler sdk (scepter.modules.utils.sampler)
|
||||
12. Probing sdk (scepter.modules.utils.probe)
|
||||
|
||||
<hr/>
|
||||
|
||||
@@ -24,7 +24,7 @@ Relies on SDKs, which are used to organize modules and SDKs that are frequently
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.config import Config
|
||||
# Initialize Config object from a dict
|
||||
fs_cfg = Config(load=False, cfg_dict={"NAME": "LocalFs"})
|
||||
print(fs_cfg.NAME)
|
||||
@@ -105,7 +105,7 @@ print(fs_cfg.args)
|
||||
Some commonly used path functions
|
||||
### Basic Usage
|
||||
```python
|
||||
from scepter.utils.directory import osp_path
|
||||
from scepter.modules.utils.directory import osp_path
|
||||
# Automatically join paths based on the path prefix
|
||||
prefix = "xxxx"
|
||||
data_file = "example_videos/1.mp4"
|
||||
@@ -114,13 +114,13 @@ print(osp_path(prefix, data_file))
|
||||
# Also outputs as xxxx/example_videos/1.mp4
|
||||
data_file = "xxxx/example_videos/1.mp4"
|
||||
print(osp_path(prefix, data_file))
|
||||
from scepter.utils.directory import get_relative_folder
|
||||
from scepter.modules.utils.directory import get_relative_folder
|
||||
# Get the folder path at a specified level according to the path
|
||||
# By default, the last level xxxx/example_videos/
|
||||
print(get_relative_folder(data_file))
|
||||
# The second last level xxxx/
|
||||
print(get_relative_folder(data_file, keep_index=-2))
|
||||
from scepter.utils.directory import get_md5
|
||||
from scepter.modules.utils.directory import get_md5
|
||||
# Get the md5 code of the text/path 34a447fb46d0b786a3999c9dad01d470
|
||||
print(get_md5(data_file))
|
||||
```
|
||||
@@ -175,8 +175,8 @@ PyTorch distributed initialization SDK. By using this SDK, users can avoid focus
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.distribute import we
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.distribute import we
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
cfg = Config(cfg_dict={}, load=False)
|
||||
|
||||
@@ -304,12 +304,12 @@ Since cloning is involved, this may cause additional GPU memory waste.
|
||||
**Returns**
|
||||
- **tensor** —— The output tensor on the CPU for process rank=0.
|
||||
|
||||
## 4. 模型导出sdk(scepter.utils.export_model)
|
||||
## 4. 模型导出sdk(scepter.modules.utils.export_model)
|
||||
APIs for exporting models to TorchScript/ONNX formats.
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.export_model import save_develop_model_multi_io
|
||||
from scepter.modules.utils.export_model import save_develop_model_multi_io
|
||||
|
||||
save_develop_model_multi_io(
|
||||
model,
|
||||
@@ -345,16 +345,16 @@ Supports importing and exporting models with multiple inputs and outputs
|
||||
**Returns**
|
||||
- **tensor** —— The output tensor on the CPU for process rank=0.
|
||||
|
||||
## 5. 文件系统sdk(scepter.utils.file_system)
|
||||
## 5. 文件系统sdk(scepter.modules.utils.file_system)
|
||||
Refer to [file_clients](file_clients.md)
|
||||
|
||||
## 6. Logging SDK(scepter.utils.logger)
|
||||
## 6. Logging SDK(scepter.modules.utils.logger)
|
||||
Used to instantiate a standard logging instance for printing information.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.logger import get_logger, init_logger
|
||||
from scepter.modules.utils.logger import get_logger, init_logger
|
||||
|
||||
std_logger = get_logger(name="scepter")
|
||||
init_logger(std_logger, log_file="", dist_launcher="pytorch")
|
||||
@@ -405,14 +405,14 @@ Calculate the time remaining until completion based on the current usage time an
|
||||
**Returns**
|
||||
- **str** —— Formatted output.
|
||||
|
||||
## 7. Video Processing SDK (scepter.utils.video_reader)
|
||||
## 7. Video Processing SDK (scepter.modules.utils.video_reader)
|
||||
APIs for handling video reading.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.utils.video_reader.video_reader import (
|
||||
from scepter.modules.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.modules.utils.video_reader.video_reader import (
|
||||
VideoReaderWrapper, EasyVideoReader, FramesReaderWrapper
|
||||
)
|
||||
```
|
||||
@@ -554,14 +554,14 @@ Iterator, with each iteration returning a tensor of a segment.
|
||||
**Returns**
|
||||
- **tensor** —— The tensor of the video segment.
|
||||
|
||||
## 8. Module Registration SDK (scepter.utils.registry)
|
||||
## 8. Module Registration SDK (scepter.modules.utils.registry)
|
||||
Used for managing various registered classes.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.registry import Registry
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.registry import Registry
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
MODELS = Registry('MODELS')
|
||||
|
||||
@@ -614,14 +614,14 @@ Register a function
|
||||
**Returns**
|
||||
- **name** —— Registration name.
|
||||
|
||||
## 9. Data SDK(scepter.utils.data)
|
||||
## 9. Data SDK(scepter.modules.utils.data)
|
||||
Used for transferring data between devices
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
from scepter.modules.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
|
||||
data = {"a": torch.Tensor([0])}
|
||||
transfer_data_to_numpy(data)
|
||||
@@ -668,7 +668,7 @@ Used for operations such as loading and evaluating models
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.model import move_model_to_cpu, load_pretrained,
|
||||
from scepter.modules.utils.model import move_model_to_cpu, load_pretrained,
|
||||
count_params, init_weights
|
||||
```
|
||||
<hr/>
|
||||
@@ -716,14 +716,14 @@ Initialize the parameters of the model modules.
|
||||
**Parameters**
|
||||
- **module** —— The torch.nn.Module model instance.
|
||||
|
||||
## 11. Sampler SDK(scepter.utils.sampler)
|
||||
## 11. Sampler SDK(scepter.modules.utils.sampler)
|
||||
Samplers are quite universal, and in most cases, custom development is not required. Here are provided several common types of sampler.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.sampler import MultiFoldDistributedSampler,
|
||||
from scepter.modules.utils.sampler import MultiFoldDistributedSampler,
|
||||
EvalDistributedSampler, MultiLevelBatchSampler, MixtureOfSamplers
|
||||
```
|
||||
<hr/>
|
||||
@@ -830,17 +830,17 @@ A sampler for multi-level indexing of large-scale data.
|
||||
|
||||
Iterator, each iteration returns an index of a sample.
|
||||
|
||||
## 12. Prober SDK(scepter.utils.probe)
|
||||
## 12. Prober SDK(scepter.modules.utils.probe)
|
||||
Used for probing variable statistics of various components.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import numpy as np
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.utils.config import Config
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.probe import ProbeData
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.probe import ProbeData
|
||||
|
||||
|
||||
class TestModel(BaseModel):
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
<h1 align="center"> Locate, Assign, Refine: Taming Customized Image Inpainting with Text-Subject Guidance </h1>
|
||||
|
||||
<p align="center">
|
||||
<strong>Yulin Pan</strong>
|
||||
·
|
||||
<strong>Chaojie Mao</strong>
|
||||
·
|
||||
<strong>Zeyinzi Jiang</strong>
|
||||
·
|
||||
<strong>Zhen Han</strong>
|
||||
·
|
||||
<strong>Jingfeng Zhang</strong>
|
||||
<br>
|
||||
<a href="https://arxiv.org/abs/2403.19534"><img src="https://img.shields.io/static/v1?label=arXiv&message=LARGen&color=red&logo=arxiv"></a>
|
||||
<a href="https://ali-vilab.github.io/largen-page/"><img src="https://img.shields.io/badge/Page-LARGen-Gree"></a>
|
||||
</p>
|
||||
|
||||
LARGen is a unified image inpainting framework that supports text-guided, subject-guided and text-subject-guided inpainting simutaneously.
|
||||
Four LARGen-based fantastic applications are now supported by SCEPTER Studio:
|
||||
1. Zoom Out
|
||||
2. Virtual Try On
|
||||
3. Text-Guided Inpainting
|
||||
4. Text-Subject-Guided Inpainting
|
||||
|
||||
## Basic Usage
|
||||
|
||||
Here's a demo showcasing the use of LARGen-based functions.
|
||||
<p align="left">
|
||||
<img src="https://raw.githubusercontent.com/ali-vilab/largen-page/main/public/images/largen.gif" width="1300">
|
||||
</p>
|
||||
|
||||
## Gallery
|
||||
|
||||
### LAR-Gen: Zoom Out
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a temple on fire</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_scene_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out1.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out2.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out3.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out4.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Virtual Try-on
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Model Image</strong></td>
|
||||
<td><strong>Model Mask</strong></td>
|
||||
<td><strong>Clothing Image</strong></td>
|
||||
<td><strong>Clothing Mask</strong></td>
|
||||
<td><strong>Try-on Output</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/virtual_try_on/model.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/ex2_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/tshirt.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/ex2_subject_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/try_on_out.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Inpainting (Text guided)
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a blue and white porcelain</td>
|
||||
<td><strong>Inpainting Mask1</strong></td>
|
||||
<td><strong>Inpainting Output1</strong></td>
|
||||
<td><strong>Inpainting Mask2</strong><br>Prompt: a clock</td>
|
||||
<td><strong>Inpainting Output2</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/inpainting_text/ex3_scene_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/ex3_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/inpainting_text.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/ex3_scene_mask2.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/inpainting_text2.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Inpainting (Text and Subject guided)
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a dog wearing sunglasses</td>
|
||||
<td><strong>Origin Mask</strong></td>
|
||||
<td><strong>Reference Image</strong></td>
|
||||
<td><strong>Reference Mask</strong></td>
|
||||
<td><strong>Inpainting Output</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_scene_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_subject_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_subject_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/inpainting_text_ref.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
## Features
|
||||
|
||||
| **Model** | **Locate** | **Assign** | **Refine** |
|
||||
|:---------:|:----------:|:----------:|:----------:|
|
||||
| SD v1.5 | ⏳ | ⏳ | ⏳ |
|
||||
| SD XL | 🪄 | 🪄 | ⏳ |
|
||||
|
||||
- 🪄 denotes that the feature has been supported.
|
||||
- ⏳ denotes that the feature has not been integrated currently.
|
||||
|
||||
|
||||
## Pretrained Models
|
||||
|
||||
| **Model** | **URL** |
|
||||
|:----------:|:-------:|
|
||||
| largen-sdxl-s22k | [ModelScope](https://www.modelscope.cn/models/iic/LARGEN/summary) |
|
||||
|
||||
|
||||
## BibTeX
|
||||
If our work is useful for your research, please consider citing:
|
||||
```bibtex
|
||||
@article{pan2024locate,
|
||||
title={Locate, Assign, Refine: Taming Customized Image Inpainting with Text-Subject Guidance},
|
||||
author={Pan, Yulin and Mao, Chaojie and Jiang, Zeyinzi and Han, Zhen and Zhang, Jingfeng},
|
||||
journal={arXiv preprint arXiv:2403.19534},
|
||||
year={2024}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,125 @@
|
||||
<p align="center">
|
||||
|
||||
<h2 align="center">SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing</h2>
|
||||
<h3 align="center">(CVPR 2024 Highlight)</h3>
|
||||
<p align="center">
|
||||
<strong>Zeyinzi Jiang</strong>
|
||||
·
|
||||
<strong>Chaojie Mao</strong>
|
||||
·
|
||||
<strong>Yulin Pan</strong>
|
||||
·
|
||||
<strong>Zhen Han</strong>
|
||||
·
|
||||
<strong>Jingfeng Zhang</strong>
|
||||
<br>
|
||||
<b>Alibaba Group</b>
|
||||
<br>
|
||||
<a href="https://arxiv.org/abs/2312.11392"><img src='https://img.shields.io/badge/arXiv-SCEdit-red' alt='Paper PDF'></a>
|
||||
<a href='https://scedit.github.io/'><img src='https://img.shields.io/badge/Project_Page-SCEdit-green' alt='Project Page'></a>
|
||||
<a href='https://github.com/modelscope/scepter'><img src='https://img.shields.io/badge/scepter-SCEdit-yellow'></a>
|
||||
<a href='https://github.com/modelscope/swift'><img src='https://img.shields.io/badge/swift-SCEdit-blue'></a>
|
||||
<br>
|
||||
</p>
|
||||
|
||||
SCEdit is an efficient generative fine-tuning framework proposed by Alibaba TongYi Vision Intelligence Lab. This framework enhances the fine-tuning capabilities for text-to-image generation downstream tasks and enables quick adaptation to specific generative scenarios, **saving 30%-50% of training memory costs compared to LoRA**. Furthermore, it can be directly extended to controllable image generation tasks, **requiring only 7.9% of the parameters that ControlNet needs for conditional generation and saving 30% of memory usage**. It supports various conditional generation tasks including edge maps, depth maps, segmentation maps, poses, color maps, and image completion.
|
||||
|
||||
## Usage
|
||||
|
||||
### Text-to-Image Generation
|
||||
```shell
|
||||
# SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i.yaml
|
||||
# SD v2.1
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd21_768_sce_t2i.yaml
|
||||
# SD XL
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i.yaml
|
||||
```
|
||||
|
||||
### Controllable Image Synthesis
|
||||
```shell
|
||||
# SD v1.5 + hed
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd15_512_sce_ctr_hed.yaml
|
||||
# SD v2.1 + canny
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml
|
||||
# SD XL + depth
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_depth.yaml
|
||||
```
|
||||
|
||||
### Gradio
|
||||
```shell
|
||||
python -m scepter.tools.webui # Then click [Use Tuners] or [Use Controller]
|
||||
```
|
||||
|
||||
## Models
|
||||
|
||||
### Model URL
|
||||
|
||||
| Model | URL |
|
||||
|--------|-------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| SCEdit | [ModelScope](https://modelscope.cn/models/iic/scepter_scedit/summary) [HuggingFace](https://huggingface.co/scepter-studio/scepter_scedit) |
|
||||
|
||||
### Text-to-Image Generation
|
||||
|
||||
| **Model** | **SCEdit** |
|
||||
|:---------:|:----------:|
|
||||
| SD 1.5 | 🪄 |
|
||||
| SD 2.1 | 🪄 |
|
||||
| SD XL | 🪄 |
|
||||
|
||||
### Controllable Image Synthesis
|
||||
|
||||
| **Model** | **Canny** | **HED** | **Depth** | **Pose** | **Color** |
|
||||
|:---------:|:---------:|:-------:|:---------:|:--------:|:---------:|
|
||||
| SD 2.1 | 🪄 | 🪄 | 🪄 | 🪄 | 🪄 |
|
||||
| SD XL | 🪄 | 🪄 | 🪄 | 🪄 | 🪄 |
|
||||
|
||||
|
||||
## Application Gallery
|
||||
|
||||
### Dragon Year Special: Dragon Tuner
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Gold Dragon Tuner</strong></td>
|
||||
<td><strong>Sloppy Dragon Tuner</strong></td>
|
||||
<td><strong>Red Dragon Tuner</strong><br> + Papercraft Mantra</td>
|
||||
<td><strong>Azure Dragon Tuner</strong><br> + Pose Control</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/scedit/tuner_gold_dragon.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/tuner_sloppy_dragon.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/tuner_mantra_papercraft_dragon.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/tuner_pose.jpeg" width="300"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### Text Effect Image
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Conditional Image</strong></td>
|
||||
<td><strong>Midas Control</strong><br>"Race track, top view"</td>
|
||||
<td><strong>Midas Control</strong><br> + Watercolor Mantra<br>"white lilies"</td>
|
||||
<td><strong>Midas Control</strong><br> + Dragon Tuner<br>"Spring Festival, Chinese dragon"</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/scedit/word_condition.png" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/word_race.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/word_lilies.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/word_festival.jpeg" width="300"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
|
||||
|
||||
## BibTeX
|
||||
|
||||
```bibtex
|
||||
@article{jiang2023scedit,
|
||||
title = {SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing},
|
||||
author = {Jiang, Zeyinzi and Mao, Chaojie and Pan, Yulin and Han, Zhen and Zhang, Jingfeng},
|
||||
year = {2023},
|
||||
journal = {arXiv preprint arXiv:2312.11392}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,116 @@
|
||||
|
||||
# StyleBooth: Image Style Editing with Multimodal Instruction
|
||||
|
||||
Zhen Han, Chaojie Mao, Zeyinzi Jiang, Yulin Pan, Jingfeng Zhang
|
||||
|
||||
Alibaba Group
|
||||
|
||||
[[paper](https://arxiv.org/abs/2404.12154)][[Model](https://modelscope.cn/models/iic/stylebooth/summary)] [[Dataset](https://modelscope.cn/models/iic/stylebooth/summary)]
|
||||
|
||||
## Abstract
|
||||
|
||||
Given an original image, image editing aims to generate an image that align with the provided instruction. The challenges are to accept multimodal inputs as instructions and a scarcity of high-quality training data, including crucial triplets of source/target image pairs and multimodal (text and image) instructions. In this paper, we focus on image style editing and present <strong>StyleBooth</strong>, a method that proposes a comprehensive framework for image editing and a feasible strategy for building a high-quality style editing dataset. We integrate encoded textual instruction and image exemplar as a unified condition for diffusion model, enabling the editing of original image following <strong>multimodal instructions</strong>. Furthermore, by <strong>iterative style-destyle tuning and editing</strong> and usability filtering, the StyleBooth dataset provides content-consistent stylized/plain image pairs in various categories of styles. To show the flexibility of StyleBooth, we conduct experiments on diverse tasks, such as textbased style editing, exemplar-based style editing and compositional style editing. The results demonstrate that the quality and variety of training data significantly enhance the ability to preserve content and improve the overall quality of generated images in editing tasks.
|
||||

|
||||
|
||||
## Gallery
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Gold Dragon Tuner</td>
|
||||
<td><strong>Graffiti Art</strong></td>
|
||||
<td><strong>Adorable Kawaii</strong></td>
|
||||
<td><strong>game-retro game</strong></td>
|
||||
<td><strong>Vincent van Gogh</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/scedit/tuner_gold_dragon.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/graffiti.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/kawaii.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/retrogame.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/vangogh.jpeg" width="240"></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong></td>
|
||||
<td><strong>Lowpoly</strong></td>
|
||||
<td><strong>Colored Pencil Art</strong></td>
|
||||
<td><strong>Watercolor</strong></td>
|
||||
<td><strong>misc-disco</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/stylebooth/mountain.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/lowpoly.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/colorpencil.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/watercolor.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/disco.jpeg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
## Features
|
||||
|
||||
| **Text-Based** | **Exemplar-Based** |
|
||||
|:--------------:|:-----------------:|
|
||||
| 🪄 | ⏳ |
|
||||
|
||||
- ✅ indicates support for both training and inference.
|
||||
- 🪄 denotes that the model has been published.
|
||||
- ⏳ denotes that the module has not been integrated currently.
|
||||
- More models will be released in the future.
|
||||
|
||||
## Run StyleBooth
|
||||
- Code implementation: See model configuration and code based on [🪄SCEPTER](https://github.com/modelscope/scepter).
|
||||
|
||||
- Demo: Try [🖥️SCEPTER Studio](https://github.com/modelscope/scepter/tree/main?tab=readme-ov-file#%EF%B8%8F-scepter-studio).
|
||||
|
||||
- Easy run:
|
||||
Try the following example script to run StyleBooth modified from [tests/modules/test_diffusion_inference.py](https://github.com/modelscope/scepter/blob/main/tests/modules/test_diffusion_inference.py):
|
||||
|
||||
```python
|
||||
# `pip install scepter>0.0.4` or
|
||||
# clone newest SCEPTER and run `PYTHONPATH=./ python <this_script>` at the main branch root.
|
||||
import os
|
||||
import unittest
|
||||
|
||||
from PIL import Image
|
||||
from torchvision.utils import save_image
|
||||
|
||||
from scepter.modules.inference.stylebooth_inference import StyleboothInference
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.logger import get_logger
|
||||
|
||||
|
||||
class DiffusionInferenceTest(unittest.TestCase):
|
||||
def setUp(self):
|
||||
print(('Testing %s.%s' % (type(self).__name__, self._testMethodName)))
|
||||
self.logger = get_logger(name='scepter')
|
||||
config_file = 'scepter/methods/studio/scepter_ui.yaml'
|
||||
cfg = Config(cfg_file=config_file)
|
||||
if 'FILE_SYSTEM' in cfg:
|
||||
for fs_info in cfg['FILE_SYSTEM']:
|
||||
FS.init_fs_client(fs_info)
|
||||
self.tmp_dir = './cache/save_data/diffusion_inference'
|
||||
if not os.path.exists(self.tmp_dir):
|
||||
os.makedirs(self.tmp_dir)
|
||||
|
||||
def tearDown(self):
|
||||
super().tearDown()
|
||||
|
||||
# uncomment this line to skip this module.
|
||||
# @unittest.skip('')
|
||||
def test_stylebooth(self):
|
||||
config_file = 'scepter/methods/studio/inference/edit/stylebooth_tb_pro.yaml'
|
||||
cfg = Config(cfg_file=config_file)
|
||||
diff_infer = StyleboothInference(logger=self.logger)
|
||||
diff_infer.init_from_cfg(cfg)
|
||||
|
||||
output = diff_infer({'prompt': 'Let this image be in the style of sai-lowpoly'},
|
||||
style_edit_image=Image.open('asset/images/inpainting_text_ref/ex4_scene_im.jpg'),
|
||||
style_guide_scale_text=7.5,
|
||||
style_guide_scale_image=1.5)
|
||||
save_path = os.path.join(self.tmp_dir,
|
||||
'stylebooth_test_lowpoly_cute_dog.png')
|
||||
save_image(output['images'], save_path)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
```
|
||||
@@ -0,0 +1,31 @@
|
||||
<h1 align="center">Dataset Management</h1>
|
||||
|
||||
SCEPTER supports three types of dataset formats: TXT, CSV, and ModelScope.
|
||||
Below are examples for each format, illustrating their details and basic usage.
|
||||
|
||||
## Modelscope Format
|
||||
|
||||
We use a [custom-stylized dataset](https://modelscope.cn/datasets/iic/style_custom_dataset/summary), which included classes 3D, anime, flat illustration, oil painting, sketch, and watercolor, each with 30 image-text pairs.
|
||||
|
||||
```python
|
||||
# pip install modelscope
|
||||
from modelscope.msdatasets import MsDataset
|
||||
ms_train_dataset = MsDataset.load('style_custom_dataset', namespace='damo', subset_name='3D', split='train_short')
|
||||
print(next(iter(ms_train_dataset)))
|
||||
```
|
||||
|
||||
## CSV Format
|
||||
|
||||
For the data format used by SCEPTER Studio, please refer to [3D_example_csv.zip](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_csv.zip) and [hed_pair.zip](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fhed_pair.zip).
|
||||
```shell
|
||||
mkdir -p cache/datasets/ && wget 'https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_csv.zip' -O cache/datasets/3D_example_csv.zip && unzip cache/datasets/3D_example_csv.zip -d cache/datasets/ && rm cache/datasets/3D_example_csv.zip
|
||||
mkdir -p cache/datasets/ && wget 'https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/hed_pair.zip' -O cache/datasets/hed_pair.zip && unzip cache/datasets/hed_pair.zip -d cache/datasets/ && rm cache/datasets/hed_pair.zip
|
||||
```
|
||||
|
||||
## TXT Format
|
||||
|
||||
To facilitate starting training in command-line mode, you can use a dataset in text format, please refer to [3D_example_txt.zip](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_txt.zip)
|
||||
|
||||
```shell
|
||||
mkdir -p cache/datasets/ && wget 'https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_txt.zip' -O cache/datasets/3D_example_txt.zip && unzip cache/datasets/3D_example_txt.zip -d cache/datasets/ && rm cache/datasets/3D_example_txt.zip
|
||||
```
|
||||
@@ -0,0 +1,45 @@
|
||||
# Inference
|
||||
|
||||
In this tutorial, we'll cover the use of the scepter framework for convenient inference, including inference using the command line or specific method classes, and we'll give examples of inference methods for additional tasks.
|
||||
|
||||
## Command Line
|
||||
Inference of SDXL generation models using the command line.
|
||||
```shell
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_xl_1024.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD XL
|
||||
```
|
||||
|
||||
## Class Instantiation
|
||||
Inference of SD2.1 generation models using the class instantiation.
|
||||
```python
|
||||
from torchvision.utils import save_image
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.logger import get_logger
|
||||
from scepter.modules.inference.diffusion_inference import DiffusionInference
|
||||
# init file system - modelscope
|
||||
FS.init_fs_client(Config(load=False, cfg_dict={'NAME': 'ModelscopeFs', 'TEMP_DIR': 'cache/data'}))
|
||||
# init model config
|
||||
logger = get_logger(name='scepter')
|
||||
cfg = Config(cfg_file='scepter/methods/studio/inference/stable_diffusion/sd21_pro.yaml')
|
||||
diff_infer = DiffusionInference(logger)
|
||||
diff_infer.init_from_cfg(cfg)
|
||||
# start inference
|
||||
output = diff_infer({'prompt': 'a cute dog'})
|
||||
save_image(output['images'], 'sd21_test_prompt_a_cute_dog.png')
|
||||
```
|
||||
|
||||
## Additional Tasks
|
||||
|
||||
### Fine-tuned Model Inference
|
||||
|
||||
```shell
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i_swift.yaml --pretrained_model 'cache/save_data/sd15_512_sce_t2i_swift/checkpoints/ldm_step-100.pth' --prompt 'A close up of a small rabbit wearing a hat and scarf' --save_folder 'trained_test_prompt_rabbit'
|
||||
```
|
||||
|
||||
### Controllable Image Synthesis Inference
|
||||
|
||||
- SCEdit
|
||||
```shell
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml --num_samples 1 --prompt 'a single flower is shown in front of a tree' --save_folder 'test_flower_canny' --image_size 768 --task control --image 'asset/images/flower.jpg' --control_mode canny --pretrained_model ms://iic/scepter_scedit@controllable_model/SD2.1/canny_control/0_SwiftSCETuning/pytorch_model.bin # canny
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml --num_samples 1 --prompt 'super mario' --save_folder 'test_mario_pose' --image_size 768 --task control --image 'asset/images/pose_source.png' --control_mode source --pretrained_model ms://iic/scepter_scedit@controllable_model/SD2.1/pose_control/0_SwiftSCETuning/pytorch_model.bin # pose
|
||||
```
|
||||
@@ -0,0 +1,84 @@
|
||||
# Training
|
||||
|
||||
We provide a framework for training and validation.
|
||||
|
||||
The scripts below are just for illustration purposes. To achieve better results, you can modify the corresponding parameters as needed.
|
||||
|
||||
## Start Training
|
||||
There are different ways to start a training:
|
||||
|
||||
- calling scepter/tools/run_train.py:
|
||||
```bash
|
||||
# calling at SCEPTER root:
|
||||
PYTHONPATH=./ python scepter/tools/run_train.py --cfg [path-to-your-yaml]
|
||||
|
||||
# calling scepter library:
|
||||
pip install scepter
|
||||
python -m scepter.tools.run_train --cfg [path-to-your-yaml]
|
||||
```
|
||||
- calling your own script:
|
||||
```bash
|
||||
# calling at SCEPTER root:
|
||||
PYTHONPATH=./ python [path-to-your-script] --cfg [path-to-your-yaml]
|
||||
|
||||
# calling scepter library:
|
||||
pip install scepter
|
||||
python [path-to-your-script] --cfg [path-to-your-yaml]
|
||||
```
|
||||
your scepter should be like:
|
||||
```python
|
||||
from scepter.tools.run_train import run
|
||||
|
||||
if __name__ == '__main__':
|
||||
run()
|
||||
```
|
||||
|
||||
## Popular Tasks
|
||||
### Text-to-Image Generation
|
||||
|
||||
- SCEdit
|
||||
```bash
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i.yaml # SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd21_768_sce_t2i.yaml # SD v2.1
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i.yaml # SD XL
|
||||
```
|
||||
|
||||
- Existing Tuning Strategies
|
||||
```bash
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_1.5_512.yaml # fully-tuning on SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_2.1_768_lora.yaml # lora-tuning on SD v2.1
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```bash
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i_datatxt.yaml
|
||||
```
|
||||
|
||||
### Controllable Image Synthesis
|
||||
|
||||
- SCEdit
|
||||
|
||||
The YAML configuration can be modified to combine different base models and conditions. The following is provided as an example.
|
||||
```bash
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd15_512_sce_ctr_hed.yaml # SD v1.5 + hed
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml # SD v2.1 + canny
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml # SD v2.1 + pose
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_depth.yaml # SD XL + depth
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color.yaml # SD XL + color
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```bash
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color_datatxt.yaml
|
||||
```
|
||||
|
||||
|
||||
## Customize Modules
|
||||
You can register your own Modules like DATASET, SAMPLERS, TRANSFORMS, MODELS, SOVLERS, HOOKS, OPTIMIZERS into SCEPTER.
|
||||
Refer to `example/`, build the modules of your task in `example/{task}`.
|
||||
```bash
|
||||
cd example/classifier
|
||||
python run.py --cfg classifier.yaml
|
||||
```
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
# Configuration file for the Sphinx documentation builder.
|
||||
#
|
||||
# This file only contains a selection of the most common options. For a full
|
||||
|
||||
@@ -15,8 +15,8 @@
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import BACKBONES
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@BACKBONES.register_class("ResNet")
|
||||
@@ -26,8 +26,8 @@ class ResNet(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import NECKS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import NECKS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@NECKS.register_class()
|
||||
@@ -37,8 +37,8 @@ class GlobalAveragePooling(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import HEADS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import HEADS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@HEADS.register_class()
|
||||
@@ -48,7 +48,7 @@ class ClassifierHead(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import LOSSES
|
||||
from scepter.modules.model.registry import LOSSES
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
@@ -60,7 +60,7 @@ class CrossEntropy(nn.Module):
|
||||
实际调用:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES, NECKS, HEADS, LOSSES, TUNERS
|
||||
from scepter.modules.model.registry import BACKBONES, NECKS, HEADS, LOSSES, TUNERS
|
||||
|
||||
backbone = BACKBONES.build(cfg.BACKBONE, logger=logger)
|
||||
neck = NECKS.build(cfg.NECK, logger=logger)
|
||||
@@ -85,8 +85,8 @@ tuner = TUNERS.build(cfg.TUNER, logger=logger)
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.model.metrics.base_metric import BaseMetric
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.base_metric import BaseMetric
|
||||
|
||||
|
||||
@METRICS.register_class("AccuracyMetric")
|
||||
@@ -97,7 +97,7 @@ class AccuracyMetric(BaseMetric):
|
||||
实际用法:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
|
||||
metric = METRICS.build(cfgs, logger)
|
||||
```
|
||||
@@ -119,8 +119,8 @@ metric = METRICS.build(cfgs, logger)
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.model.tokenizers import BaseTokenizer
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.tokenizers import BaseTokenizer
|
||||
|
||||
|
||||
@TOKENIZERS.register_class()
|
||||
@@ -131,7 +131,7 @@ class BaseBertTokenizer(BaseTokenizer):
|
||||
实际用法:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
|
||||
tokenizer = TOKENIZERS.build(cfgs, logger)
|
||||
```
|
||||
@@ -149,8 +149,8 @@ tokenizer = TOKENIZERS.build(cfgs, logger)
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.model.networks.train_module import TrainModule
|
||||
from scepter.modules.model.registry import MODELS
|
||||
from scepter.modules.model.networks.train_module import TrainModule
|
||||
|
||||
|
||||
@MODELS.register_class()
|
||||
@@ -161,7 +161,7 @@ class Classifier(TrainModule):
|
||||
实际用法:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.modules.model.registry import MODELS
|
||||
|
||||
model = MODELS.build(self.cfg.MODEL, logger=self.logger)
|
||||
```
|
||||
|
||||
@@ -9,8 +9,8 @@
|
||||
子lr_schedulers继承时用法:
|
||||
|
||||
```python
|
||||
from scepter.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
from scepter.modules.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.modules.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
|
||||
|
||||
@LR_SCHEDULERS.register_class()
|
||||
@@ -48,8 +48,8 @@ lr_schedulers的基类,支持注册操作,可根据需要自定义;
|
||||
子optimizers继承时用法:
|
||||
|
||||
```python
|
||||
from scepter.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.opt.optimizers.registry import OPTIMIZERS
|
||||
from scepter.modules.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.modules.opt.optimizers.registry import OPTIMIZERS
|
||||
|
||||
|
||||
@OPTIMIZERS.register_class()
|
||||
|
||||
@@ -6,10 +6,10 @@
|
||||
|
||||
支持3类文件IO Handler:
|
||||
|
||||
1. scepter.utils.file_clients.AliyunOssFs
|
||||
2. scepter.utils.file_clients.LocalFs
|
||||
3. scepter.utils.file_clients.HttpFs
|
||||
4. scepter.utils.file_clients.ModelscopeFs
|
||||
1. scepter.modules.utils.file_clients.AliyunOssFs
|
||||
2. scepter.modules.utils.file_clients.LocalFs
|
||||
3. scepter.modules.utils.file_clients.HttpFs
|
||||
4. scepter.modules.utils.file_clients.ModelscopeFs
|
||||
|
||||
|
||||
<hr/>
|
||||
@@ -17,8 +17,8 @@
|
||||
## 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
fs_cfg = Config(load=False, cfg_dict={
|
||||
"NAME": "AliyunOssFs",
|
||||
|
||||
@@ -3,18 +3,18 @@
|
||||
依赖SDK,该部分用于对框架全局经常复用的模块和sdk进行整理,并根据功能相关性进行聚合。
|
||||
|
||||
## 总览
|
||||
1. 参数sdk(scepter.utils.config)
|
||||
2. 路径sdk(scepter.utils.directory)
|
||||
3. torch分布式sdk(scepter.utils.distribute)
|
||||
4. 模型导出sdk(scepter.utils.export_model)
|
||||
5. 文件系统sdk(scepter.utils.file_system)
|
||||
6. 日志sdk(scepter.utils.logger)
|
||||
7. 视频处理sdk(scepter.utils.video_reader),文档参考(video_reader.md)
|
||||
8. 模块注册sdk(scepter.utils.registry)
|
||||
9. 数据sdk(scepter.utils.data)
|
||||
10. 模型sdk(scepter.utils.model)
|
||||
11. 采样器sdk(scepter.utils.sampler)
|
||||
12. 探针器sdk(scepter.utils.probe)
|
||||
1. 参数sdk(scepter.modules.utils.config)
|
||||
2. 路径sdk(scepter.modules.utils.directory)
|
||||
3. torch分布式sdk(scepter.modules.utils.distribute)
|
||||
4. 模型导出sdk(scepter.modules.utils.export_model)
|
||||
5. 文件系统sdk(scepter.modules.utils.file_system)
|
||||
6. 日志sdk(scepter.modules.utils.logger)
|
||||
7. 视频处理sdk(scepter.modules.utils.video_reader),文档参考(video_reader.md)
|
||||
8. 模块注册sdk(scepter.modules.utils.registry)
|
||||
9. 数据sdk(scepter.modules.utils.data)
|
||||
10. 模型sdk(scepter.modules.utils.model)
|
||||
11. 采样器sdk(scepter.modules.utils.sampler)
|
||||
12. 探针器sdk(scepter.modules.utils.probe)
|
||||
|
||||
<hr/>
|
||||
|
||||
@@ -23,7 +23,7 @@
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
# 从一个dict对象 初始化 Config对象
|
||||
fs_cfg = Config(load=False, cfg_dict={"NAME": "LocalFs"})
|
||||
@@ -97,7 +97,7 @@ print(fs_cfg.args)
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.directory import osp_path
|
||||
from scepter.modules.utils.directory import osp_path
|
||||
|
||||
# 根据路径前缀进行自动化路径拼接
|
||||
prefix = "xxxx"
|
||||
@@ -108,7 +108,7 @@ print(osp_path(prefix, data_file))
|
||||
data_file = "xxxx/example_videos/1.mp4"
|
||||
print(osp_path(prefix, data_file))
|
||||
|
||||
from scepter.utils.directory import get_relative_folder
|
||||
from scepter.modules.utils.directory import get_relative_folder
|
||||
|
||||
# 根据路径获取指定层级的文件夹路径
|
||||
# 默认最后一级 xxxx/example_videos/
|
||||
@@ -116,7 +116,7 @@ print(get_relative_folder(data_file))
|
||||
# 倒数第二级 xxxx/
|
||||
print(get_relative_folder(data_file, keep_index=-2))
|
||||
|
||||
from scepter.utils.directory import get_md5
|
||||
from scepter.modules.utils.directory import get_md5
|
||||
|
||||
# 获取文本/路径的md5码 34a447fb46d0b786a3999c9dad01d470
|
||||
print(get_md5(data_file))
|
||||
@@ -172,8 +172,8 @@ torch分布式初始化sdk,使用该sdk,可以让用户不要关注torch的
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.distribute import we
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.distribute import we
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
cfg = Config(cfg_dict={}, load=False)
|
||||
|
||||
@@ -304,12 +304,12 @@ we.init_env(cfg, fn, logger=None)
|
||||
**Returns**
|
||||
- **tensor** —— 输出的在进程rank=0上的cpu的tensor。
|
||||
|
||||
## 4. 模型导出sdk(scepter.utils.export_model)
|
||||
## 4. 模型导出sdk(scepter.modules.utils.export_model)
|
||||
用于模型导出为torchscript/Onnx格式的api。
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.export_model import save_develop_model_multi_io
|
||||
from scepter.modules.utils.export_model import save_develop_model_multi_io
|
||||
|
||||
save_develop_model_multi_io(
|
||||
model,
|
||||
@@ -347,16 +347,16 @@ input_type 一一对应。
|
||||
**Returns**
|
||||
- **tensor** —— 输出的在进程rank=0上的cpu的tensor。
|
||||
|
||||
## 5. 文件系统sdk(scepter.utils.file_system)
|
||||
## 5. 文件系统sdk(scepter.modules.utils.file_system)
|
||||
参考[file_clients](file_clients.md)
|
||||
|
||||
## 6. 日志sdk(scepter.utils.logger)
|
||||
## 6. 日志sdk(scepter.modules.utils.logger)
|
||||
用于实例化一个标准的日志实例,用于打印信息。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.logger import get_logger, init_logger
|
||||
from scepter.modules.utils.logger import get_logger, init_logger
|
||||
|
||||
std_logger = get_logger(name="scepter")
|
||||
init_logger(std_logger, log_file="", dist_launcher="pytorch")
|
||||
@@ -407,14 +407,14 @@ init_logger(std_logger, log_file="", dist_launcher="pytorch")
|
||||
**Returns**
|
||||
- **str** —— 格式化的输出。
|
||||
|
||||
## 7. 视频处理sdk(scepter.utils.video_reader)
|
||||
## 7. 视频处理sdk(scepter.modules.utils.video_reader)
|
||||
用于处理视频读取的api。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.utils.video_reader.video_reader import (
|
||||
from scepter.modules.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.modules.utils.video_reader.video_reader import (
|
||||
VideoReaderWrapper, EasyVideoReader, FramesReaderWrapper
|
||||
)
|
||||
```
|
||||
@@ -556,14 +556,14 @@ overlap: Union[float, Fraction, str] = Fraction(0), transforms: Optional[Callabl
|
||||
**Returns**
|
||||
- **tensor** —— 视频片段的tensor。
|
||||
|
||||
## 8. 模块注册sdk(scepter.utils.registry)
|
||||
## 8. 模块注册sdk(scepter.modules.utils.registry)
|
||||
用于管理各种注册的类。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.registry import Registry
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.registry import Registry
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
MODELS = Registry('MODELS')
|
||||
|
||||
@@ -616,14 +616,14 @@ build目标类的实例
|
||||
**Returns**
|
||||
- **name** —— 注册名称。
|
||||
|
||||
## 9. 数据sdk(scepter.utils.data)
|
||||
## 9. 数据sdk(scepter.modules.utils.data)
|
||||
用于数据在设备间转移
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
from scepter.modules.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
|
||||
data = {"a": torch.Tensor([0])}
|
||||
transfer_data_to_numpy(data)
|
||||
@@ -670,7 +670,7 @@ transfer_data_to_cuda(data)
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.model import move_model_to_cpu, load_pretrained,
|
||||
from scepter.modules.utils.model import move_model_to_cpu, load_pretrained,
|
||||
count_params, init_weights
|
||||
```
|
||||
<hr/>
|
||||
@@ -718,14 +718,14 @@ from scepter.utils.model import move_model_to_cpu, load_pretrained,
|
||||
**Parameters**
|
||||
- **module** —— torch.nn.Module模型实例。
|
||||
|
||||
## 11. 采样器sdk(scepter.utils.sampler)
|
||||
## 11. 采样器sdk(scepter.modules.utils.sampler)
|
||||
采样器比较具有通用性,大多数情况下不会进行定制开发,这里提供了几类常用的sampler采样器。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.sampler import MultiFoldDistributedSampler,
|
||||
from scepter.modules.utils.sampler import MultiFoldDistributedSampler,
|
||||
EvalDistributedSampler, MultiLevelBatchSampler, MixtureOfSamplers
|
||||
```
|
||||
<hr/>
|
||||
@@ -832,17 +832,17 @@ from scepter.utils.sampler import MultiFoldDistributedSampler,
|
||||
|
||||
迭代器,每迭代一次得到一个样本的index
|
||||
|
||||
## 12. 探针器sdk(scepter.utils.probe)
|
||||
## 12. 探针器sdk(scepter.modules.utils.probe)
|
||||
用于探针各个组件的变量统计
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
import numpy as np
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.utils.config import Config
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.probe import ProbeData
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.probe import ProbeData
|
||||
|
||||
|
||||
class TestModel(BaseModel):
|
||||
|
||||
@@ -6,4 +6,5 @@ dependencies:
|
||||
- pip>=20.3
|
||||
- numpy>=1.23.1
|
||||
- pip:
|
||||
- -r requirements/recommended.txt
|
||||
- -r requirements.txt
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import numpy as np
|
||||
import torchvision
|
||||
|
||||
from scepter.modules.data.dataset.base_dataset import BaseDataset
|
||||
from scepter.modules.data.dataset.registry import DATASETS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
|
||||
@@ -8,19 +8,18 @@
|
||||
<a href="https://github.com/modelscope/scepter/"><img src="https://img.shields.io/badge/scepter-Build from source-6FEBB9.svg"></a>
|
||||
</p>
|
||||
|
||||
## 📖 Table of Contents
|
||||
- [News](#-news)
|
||||
- [Introduction](#-introduction)
|
||||
- [Installation](#%EF%B8%8F-installation)
|
||||
- [Getting Started](#-getting-started)
|
||||
- [SCEPTER Studio](#%EF%B8%8F-scepter-studio)
|
||||
- [Gallery](#%EF%B8%8F-gallery)
|
||||
- [Features](#-features)
|
||||
- [Learn More](#-learn-more)
|
||||
- [License](#license)
|
||||
- [Acknowledgement](#acknowledgement)
|
||||
🪄SCEPTER is an open-source code repository dedicated to generative training, fine-tuning, and inference, encompassing a suite of downstream tasks such as image generation, transfer, editing.
|
||||
SCEPTER integrates popular community-driven implementations as well as proprietary methods by Tongyi Lab of Alibaba Group, offering a comprehensive toolkit for researchers and practitioners in the field of AIGC. This versatile library is designed to facilitate innovation and accelerate development in the rapidly evolving domain of generative models.
|
||||
|
||||
SCEPTER offers 3 core components:
|
||||
- [Generative training and inference framework](#tutorials)
|
||||
- [Easy implementation of popular approaches](#currently-supported-approaches)
|
||||
- [Interactive user interface: SCEPTER Studio](#launch)
|
||||
|
||||
|
||||
## 🎉 News
|
||||
- [2024.05]: Introducing SCEPTER v1, supporting customized image edit tasks! Simply provide 10 image pairs, SCEPTER will tune an edit tuner for your own Image-to-Image tasks, like `Clay Style`, `De-Text`, `Segmentation`, etc.
|
||||
- [2024.04]: New [StyleBooth](https://ali-vilab.github.io/stylebooth-page/) demo on SCEPTER Studio for`Text-Based Style Editing`.
|
||||
- [2024.03]: We optimize the training UI and checkpoint management. New [LAR-Gen](https://arxiv.org/abs/2403.19534) model has been added on SCEPTER Studio, supporting `zoom-out`, `virtual try on`, `inpainting`.
|
||||
- [2024.02]: We release new SCEdit controllable image synthesis models for SD v2.1 and SD XL. Multiple strategies applied to accelerate inference time for SCEPTER Studio.
|
||||
- [2024.01]: We release **SCEPTER Studio**, an integrated toolkit for data management, model training and inference based on [Gradio](https://www.gradio.app/).
|
||||
@@ -28,152 +27,91 @@
|
||||
- [2023.12]: We propose [SCEdit](https://arxiv.org/abs/2312.11392), an efficient and controllable generation framework.
|
||||
- [2023.12]: We release [🪄SCEPTER](https://github.com/modelscope/scepter/) library.
|
||||
|
||||
## 📝 Introduction
|
||||
|
||||
SCEPTER is an open-source code repository dedicated to generative training, fine-tuning, and inference, encompassing a suite of downstream tasks such as image generation, transfer, editing. It integrates popular community-driven implementations as well as proprietary methods by Tongyi Lab of Alibaba Group, offering a comprehensive toolkit for researchers and practitioners in the field of AIGC. This versatile library is designed to facilitate innovation and accelerate development in the rapidly evolving domain of generative models.
|
||||
## 🖼 Gallery for Recent Works
|
||||
|
||||
Main Feature:
|
||||
### Edit Tuners
|
||||
|
||||
- Task:
|
||||
- Text-to-image generation
|
||||
- Controllable image synthesis
|
||||
- Image editing
|
||||
- Training / Inference:
|
||||
- Distribute: DDP / FSDP / FairScale / Xformers
|
||||
- File system: Local / Http / OSS / Modelscope
|
||||
- Deploy:
|
||||
- Data management
|
||||
- Training
|
||||
- Inference
|
||||
Simply provide 10 image pairs, SCEPTER will tune an edit tuner for your own Image-to-Image tasks, like `Clay Style`, `De-Text`, `Segmentation`, etc.
|
||||
Try our official few-shot datasets: [De-Text](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fdetext.zip), [Image2Hed](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fhed_pair.zip), [Image2Depth](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fimage2depth.zip), [Depth2Image](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fdepth2image.zip).
|
||||
|
||||
Currently supported approaches (and counting):
|
||||
|
||||
1. SD Series: [Stable Diffusion v1.5](https://huggingface.co/runwayml/stable-diffusion-v1-5) / [Stable Diffusion v2.1](https://huggingface.co/runwayml/stable-diffusion-v1-5) / [Stable Diffusion XL](https://huggingface.co/stabilityai/stable-diffusion-xl-base-1.0)
|
||||
2. SCEdit(CVPR2024): [SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing](https://arxiv.org/abs/2312.11392) [](https://arxiv.org/abs/2312.11392) [](https://scedit.github.io/)
|
||||
3. Res-Tuning(NeurIPS2023 TODO): [Res-Tuning: A Flexible and Efficient Tuning Paradigm via Unbinding Tuner from Backbone](https://arxiv.org/abs/2310.19859) [](https://arxiv.org/abs/2310.19859) [](https://res-tuning.github.io/)
|
||||
4. LAR-Gen: [Locate, Assign, Refine: Taming Customized Image Inpainting with Text-Subject Guidance](https://arxiv.org/abs/2403.19534) [](https://arxiv.org/abs/2403.19534) [](https://ali-vilab.github.io/largen-page/)
|
||||
<table><tbody>
|
||||
<tr>
|
||||
<th align="center" colspan="4">Clay Style<br>Prompt: "Convert this image into clay style"</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/edit_tuner/vermeer.jpeg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/clay_vermeer.jpeg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/cat_512.jpg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/clay_cat.jpeg" width="300"></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<th align="center" colspan="2">De-Text<br>Prompt: "Remove the texts"</th>
|
||||
<th align="center" colspan="2">Image2Hed<br>Prompt: "Convert to an edge map"</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/edit_tuner/text.jpg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/detext.jpeg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/cat_512.jpg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/hed.jpeg" width="300"></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<th align="center" colspan="2">Image2Depth<br>Prompt: "Calculate the depth map"</th>
|
||||
<th align="center" colspan="2">Depth2Image<br>Prompt: "Convert depth map into color image"</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/edit_tuner/house.jpg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/image2depth.jpeg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/depth.jpg" width="300"></td>
|
||||
<td><img src="asset/images/edit_tuner/depth2image.jpeg" width="300"></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
Note: Left image is input and right image is output.
|
||||
|
||||
## 🛠️ Installation
|
||||
|
||||
- Create new environment
|
||||
- Create new environment with `conda` command:
|
||||
|
||||
```shell
|
||||
conda env create -f environment.yaml
|
||||
conda activate scepter
|
||||
```
|
||||
- We recommend installing the specific version of PyTorch and accelerate toolbox [xFormers](https://pypi.org/project/xformers/). You can install these recommended version by pip:
|
||||
|
||||
- Install with `pip` command:
|
||||
|
||||
We recommend installing the specific version of PyTorch and accelerate toolbox [xFormers](https://pypi.org/project/xformers/). You can install these recommended version by pip:
|
||||
```shell
|
||||
pip install -r requirements/recommended.txt
|
||||
```
|
||||
|
||||
- Install SCEPTER by the `pip` command:
|
||||
|
||||
```shell
|
||||
pip install scepter
|
||||
```
|
||||
|
||||
## 🚀 Getting Started
|
||||
## 🧩 Generative Framework
|
||||
|
||||
### Dataset
|
||||
### Tutorials
|
||||
|
||||
#### Modelscope Format
|
||||
| Documentation | Key Features |
|
||||
|:---------------------------------------------------|:----------------------------------|
|
||||
| [Train](docs/en/tutorials/train.md) | DDP / FSDP / FairScale / Xformers |
|
||||
| [Inference](docs/en/tutorials/inference.md) | Dynamic load/unload |
|
||||
| [Dataset Management](docs/en/tutorials/dataset.md) | Local / Http / OSS / Modelscope |
|
||||
|
||||
We use a [custom-stylized dataset](https://modelscope.cn/datasets/damo/style_custom_dataset/summary), which included classes 3D, anime, flat illustration, oil painting, sketch, and watercolor, each with 30 image-text pairs.
|
||||
|
||||
```python
|
||||
# pip install modelscope
|
||||
from modelscope.msdatasets import MsDataset
|
||||
ms_train_dataset = MsDataset.load('style_custom_dataset', namespace='damo', subset_name='3D', split='train_short')
|
||||
print(next(iter(ms_train_dataset)))
|
||||
```
|
||||
## 📝 Popular Approaches
|
||||
|
||||
#### CSV Format
|
||||
### Currently supported approaches
|
||||
|
||||
For the data format used by SCEPTER Studio, please refer to [3D_example_csv.zip](https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=datasets/3D_example_csv.zip).
|
||||
|
||||
#### TXT Format
|
||||
|
||||
To facilitate starting training in command-line mode, you can use a dataset in text format, please refer to [3D_example_txt.zip](https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=datasets/3D_example_txt.zip)
|
||||
|
||||
```shell
|
||||
mkdir -p cache/datasets/ && wget 'https://modelscope.cn/api/v1/models/damo/scepter_scedit/repo?Revision=master&FilePath=dataset/3D_example_txt.zip' -O cache/datasets/3D_example_txt.zip && unzip cache/datasets/3D_example_txt.zip -d cache/datasets/ && rm cache/datasets/3D_example_txt.zip
|
||||
```
|
||||
|
||||
### Training
|
||||
|
||||
We provide a framework for training and inference, so the script below is just for illustration purposes. To achieve better results, you can modify the corresponding parameters as needed.
|
||||
|
||||
#### Text-to-Image Generation
|
||||
|
||||
- SCEdit
|
||||
```python
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i.yaml # SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd21_768_sce_t2i.yaml # SD v2.1
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i.yaml # SD XL
|
||||
```
|
||||
|
||||
- Existing Tuning Strategies
|
||||
```python
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_1.5_512.yaml # fully-tuning on SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_2.1_768_lora.yaml # lora-tuning on SD v2.1
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```python
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i_datatxt.yaml
|
||||
```
|
||||
|
||||
#### Controllable Image Synthesis
|
||||
|
||||
- SCEdit
|
||||
|
||||
The YAML configuration can be modified to combine different base models and conditions. The following is provided as an example.
|
||||
```python
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd15_512_sce_ctr_hed.yaml # SD v1.5 + hed
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml # SD v2.1 + canny
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml # SD v2.1 + pose
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_depth.yaml # SD XL + depth
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color.yaml # SD XL + color
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```python
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color_datatxt.yaml
|
||||
```
|
||||
|
||||
### Inference
|
||||
|
||||
#### Base Model Inference
|
||||
|
||||
```python
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_1.5_512.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD v1.5
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_2.1_768.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD v2.1
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_xl_1024.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD XL
|
||||
```
|
||||
|
||||
#### Fine-tuned Model Inference
|
||||
|
||||
```python
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i_swift.yaml --pretrained_model 'cache/save_data/sd15_512_sce_t2i_swift/checkpoints/ldm_step-100.pth' --prompt 'A close up of a small rabbit wearing a hat and scarf' --save_folder 'trained_test_prompt_rabbit'
|
||||
```
|
||||
|
||||
#### Controllable Image Synthesis Inference
|
||||
|
||||
- SCEdit
|
||||
```python
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml --num_samples 1 --prompt 'a single flower is shown in front of a tree' --save_folder 'test_flower_canny' --image_size 768 --task control --image 'asset/images/flower.jpg' --control_mode canny --pretrained_model ms://damo/scepter_scedit@controllable_model/SD2.1/canny_control/0_SwiftSCETuning/pytorch_model.bin # canny
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml --num_samples 1 --prompt 'super mario' --save_folder 'test_mario_pose' --image_size 768 --task control --image 'asset/images/pose_source.png' --control_mode source --pretrained_model ms://damo/scepter_scedit@controllable_model/SD2.1/pose_control/0_SwiftSCETuning/pytorch_model.bin # pose
|
||||
```
|
||||
|
||||
### Customize Modules
|
||||
Refer to `example`, build the modules of your task in `example/{task}`.
|
||||
```python
|
||||
cd example/classifier
|
||||
python run.py --cfg classifier.yaml
|
||||
```
|
||||
| Tasks | Methods | Links |
|
||||
|:----------------------------:|:--------------------------------------------:|:------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Text-to-image generation | SD v1.5 | [](https://huggingface.co/runwayml/stable-diffusion-v1-5) |
|
||||
| Text-to-image generation | SD v2.1 | [](https://huggingface.co/runwayml/stable-diffusion-v1-5) |
|
||||
| Text-to-image generation | SD-XL | [](https://huggingface.co/stabilityai/stable-diffusion-xl-base-1.0) |
|
||||
| Efficient Tuning | LoRA | [](https://arxiv.org/abs/2106.09685) |
|
||||
| Efficient Tuning | Res-Tuning(NeurIPS23) | [](https://arxiv.org/abs/2310.19859) [](https://res-tuning.github.io/) |
|
||||
| Controllable image synthesis | [🌟SCEdit(CVPR24)](docs/en/tasks/scedit.md) | [](https://arxiv.org/abs/2312.11392) [](https://scedit.github.io/) |
|
||||
| Image editing | [🌟LAR-Gen](docs/en/tasks/largen.md) | [](https://arxiv.org/abs/2403.19534) [](https://ali-vilab.github.io/largen-page/) |
|
||||
| Image editing | [🌟StyleBooth](docs/en/tasks/stylebooth.md) | [](https://arxiv.org/abs/2404.12154) [](https://ali-vilab.github.io/stylebooth-page/) |
|
||||
|
||||
|
||||
## 🖥️ SCEPTER Studio
|
||||
@@ -196,163 +134,15 @@ The startup of **SCEPTER Studio** eliminates the need for manual downloading and
|
||||
Depending on the network and hardware situation, the initial startup usually requires 15-60 minutes, primarily involving the download and processing of SDv1.5, SDv2.1, and SDXL models.
|
||||
Therefore, subsequent startups will become much faster (about one minute) as downloading is no longer required.
|
||||
|
||||
* LAR-Gen: we release `zoom-out`, `virtual try on`, `inpainting(text guided)`, `inpainting(text + reference image guided)` image editing capabilities.
|
||||
Please note that the **Data Preprocess** button must be clicked before clicking the **Generate** button.
|
||||
<p align="center">
|
||||
<img src="https://raw.githubusercontent.com/ali-vilab/largen-page/main/public/images/largen.gif">
|
||||
</p>
|
||||
### Usage Demo
|
||||
|
||||
### Modelscope Studio
|
||||
| [Image Editing](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fimage_editing_20240419.webm) | [Training](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Ftraining_20240419.webm) | [Model Sharing](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_sharing_20240419.webm) | [Model Inference](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_inference_20240419.webm) | [Data Management](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fdata_management_20240419.webm) |
|
||||
|:----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:--------------------------------------------:|
|
||||
| <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fimage_editing_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Ftraining_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_sharing_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_inference_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fdata_management_20240419.webm" width="240" controls></video> |
|
||||
|
||||
We deploy a work studio on Modelscope that includes only the inference tab, please refer to [ms_scepter_studio](https://www.modelscope.cn/studios/damo/scepter_studio/summary)
|
||||
### Modelscope Studio & Huggingface Space
|
||||
|
||||
## 🖼️ Gallery
|
||||
|
||||
### LAR-Gen: Zoom Out
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a temple on fire</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/zoom_out/ex1_scene_im.jpg" width="240"></td>
|
||||
<td><img src="asset/images/zoom_out/ex1_zoom_out1.jpg" width="240"></td>
|
||||
<td><img src="./asset/images/zoom_out/ex1_zoom_out2.jpg" width="240"></td>
|
||||
<td><img src="./asset/images/zoom_out/ex1_zoom_out3.jpg" width="240"></td>
|
||||
<td><img src="./asset/images/zoom_out/ex1_zoom_out4.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Virtual Try-on
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Model Image</strong></td>
|
||||
<td><strong>Model Mask</strong></td>
|
||||
<td><strong>Clothing Image</strong></td>
|
||||
<td><strong>Clothing Mask</strong></td>
|
||||
<td><strong>Try-on Output</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/virtual_try_on/model.jpg" width="240"></td>
|
||||
<td><img src="asset/images/virtual_try_on/ex2_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="asset/images/virtual_try_on/tshirt.jpg" width="240"></td>
|
||||
<td><img src="asset/images/virtual_try_on/ex2_subject_mask.jpg" width="240"></td>
|
||||
<td><img src="asset/images/virtual_try_on/try_on_out.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Inpainting (Text guided)
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a blue and white porcelain</td>
|
||||
<td><strong>Inpainting Mask1</strong></td>
|
||||
<td><strong>Inpainting Output1</strong></td>
|
||||
<td><strong>Inpainting Mask2</strong><br>Prompt: a clock</td>
|
||||
<td><strong>Inpainting Output2</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/inpainting_text/ex3_scene_im.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text/ex3_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text/inpainting_text.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text/ex3_scene_mask2.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text/inpainting_text2.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Inpainting (Text and Subject guided)
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a dog wearing sunglasses</td>
|
||||
<td><strong>Origin Mask</strong></td>
|
||||
<td><strong>Reference Image</strong></td>
|
||||
<td><strong>Reference Mask</strong></td>
|
||||
<td><strong>Inpainting Output</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="asset/images/inpainting_text_ref/ex4_scene_im.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text_ref/ex4_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text_ref/ex4_subject_im.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text_ref/ex4_subject_mask.jpg" width="240"></td>
|
||||
<td><img src="asset/images/inpainting_text_ref/inpainting_text_ref.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### Dragon Year Special: Dragon Tuner
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Gold Dragon Tuner</strong></td>
|
||||
<td><strong>Sloppy Dragon Tuner</strong></td>
|
||||
<td><strong>Red Dragon Tuner</strong><br> + Papercraft Mantra</td>
|
||||
<td><strong>Azure Dragon Tuner</strong><br> + Pose Control</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/tuner_gold_dragon.jpeg?raw=true" width="300"></td>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/tuner_sloppy_dragon.jpeg?raw=true" width="300"></td>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/tuner_mantra_papercraft_dragon.jpeg?raw=true" width="300"></td>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/tuner_pose.jpeg?raw=true" width="300"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### Text Effect Image
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Conditional Image</strong></td>
|
||||
<td><strong>Midas Control</strong><br>"Race track, top view"</td>
|
||||
<td><strong>Midas Control</strong><br> + Watercolor Mantra<br>"white lilies"</td>
|
||||
<td><strong>Midas Control</strong><br> + Dragon Tuner<br>"Spring Festival, Chinese dragon"</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/word_condition.png?raw=true" width="300"></td>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/word_race.jpeg?raw=true" width="300"></td>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/word_lilies.jpeg?raw=true" width="300"></td>
|
||||
<td><img src="https://github.com/hanzhn/datas/blob/main/scepter/readme/word_festival.jpeg?raw=true" width="300"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
## ✨ Features
|
||||
|
||||
### Text-to-Image Generation
|
||||
|
||||
| **Model** | **SCEdit** | **Full** | **LoRA** |
|
||||
|:---------:|:----------:|:--------:|:--------:|
|
||||
| SD 1.5 | 🪄 | ✅ | ✅ |
|
||||
| SD 2.1 | 🪄 | ✅ | ✅ |
|
||||
| SD XL | 🪄 | ✅ | ✅ |
|
||||
|
||||
### Controllable Image Synthesis
|
||||
- SCEdit
|
||||
|
||||
| **Model** | **Canny** | **HED** | **Depth** | **Pose** | **Color** |
|
||||
|:---------:|:---------:|:-------:|:---------:|:--------:|:---------:|
|
||||
| SD 1.5 | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| SD 2.1 | 🪄 | 🪄 | 🪄 | 🪄 | 🪄 |
|
||||
| SD XL | 🪄 | 🪄 | 🪄 | 🪄 | 🪄 |
|
||||
|
||||
### Image Editing
|
||||
- LAR-Gen
|
||||
|
||||
| **Model** | **Locate** | **Assign** | **Refine** |
|
||||
|:---------:|:----------:|:----------:|:----------:|
|
||||
| SD XL | 🪄 | 🪄 | ⏳ |
|
||||
|
||||
### Model URL
|
||||
|
||||
- ✅ indicates support for both training and inference.
|
||||
- 🪄 denotes that the model has been published.
|
||||
- ⏳ denotes that the module has not been integrated currently.
|
||||
- More models will be released in the future.
|
||||
|
||||
| Model | URL |
|
||||
|--------|------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| SCEdit | [ModelScope](https://modelscope.cn/models/iic/scepter_scedit/summary) [HuggingFace](https://huggingface.co/scepter-studio/scepter_scedit) |
|
||||
| LAR-Gen | [ModelScope](https://www.modelscope.cn/models/iic/LARGEN/summary) |
|
||||
|
||||
PS: Scripts running within the SCEPTER framework will automatically fetch and load models based on the required dependency files, eliminating the need for manual downloads.
|
||||
We deploy a work studio on Modelscope that includes only the inference tab, please refer to [ms_scepter_studio](https://www.modelscope.cn/studios/iic/scepter_studio/summary) and [hf_scepter_studio](https://huggingface.co/spaces/modelscope/scepter_studio)
|
||||
|
||||
|
||||
## 🔍 Learn More
|
||||
@@ -369,6 +159,7 @@ PS: Scripts running within the SCEPTER framework will automatically fetch and lo
|
||||
|
||||
SWIFT (Scalable lightWeight Infrastructure for Fine-Tuning) is an extensible framwork designed to faciliate lightweight model fine-tuning and inference.
|
||||
|
||||
|
||||
## BibTeX
|
||||
If our work is useful for your research, please consider citing:
|
||||
```bibtex
|
||||
@@ -384,5 +175,6 @@ If our work is useful for your research, please consider citing:
|
||||
|
||||
This project is licensed under the [Apache License (Version 2.0)](https://github.com/modelscope/modelscope/blob/master/LICENSE).
|
||||
|
||||
|
||||
## Acknowledgement
|
||||
Thanks to [Stability-AI](https://github.com/Stability-AI), [SWIFT library](https://github.com/modelscope/swift/) and [Fooocus](https://github.com/lllyasviel/Fooocus) for their awesome work.
|
||||
|
||||
@@ -2,7 +2,7 @@ albumentations
|
||||
bezier
|
||||
einops
|
||||
modelscope
|
||||
ms-swift>=1.5.2
|
||||
ms-swift>=2.0.1
|
||||
numpy
|
||||
open_clip_torch
|
||||
opencv-python
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
bitsandbytes
|
||||
gradio>=3.47.1,<4.0.0
|
||||
imagehash
|
||||
psutil
|
||||
tiktoken
|
||||
transformers_stream_generator
|
||||
|
||||
@@ -0,0 +1,250 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/edit_512_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionEdit
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://iic/stylebooth@models/stylebooth-tb-5000-0.bin
|
||||
IGNORE_KEYS: [ ]
|
||||
CONCAT_NO_SCALE_FACTOR: True
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 8
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: #7.5
|
||||
image: 1.5
|
||||
text: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: cache/datasets/hed_pair
|
||||
MS_DATASET_NAMESPACE: ""
|
||||
MS_DATASET_SPLIT: "train"
|
||||
MS_DATASET_SUBNAME: ""
|
||||
PROMPT_PREFIX: ""
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFileList
|
||||
FILE_KEYS: ['img_path', 'src_path']
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'image', 'condition_cat' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'condition_cat', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
|
||||
EVAL_DATA:
|
||||
NAME: Text2ImageDataset
|
||||
MODE: eval
|
||||
PROMPT_FILE:
|
||||
PROMPT_DATA: [ "Convert to an edge map#;#cache/datasets/hed_pair/images/src_001.jpeg" ]
|
||||
IMAGE_SIZE: [ 512, 512 ]
|
||||
FIELDS: [ "prompt", "src_path" ]
|
||||
DELIMITER: '#;#'
|
||||
PROMPT_PREFIX: ''
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFileList
|
||||
FILE_KEYS: [ 'src_path' ]
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'condition_cat' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'condition_cat', 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
SAVE_PROBE_PREFIX: 'image'
|
||||
@@ -0,0 +1,234 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd15_512_textlora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$"
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-v1-5@v1-5-pruned-emaonly.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
USE_GRAD: True
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 512
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 512
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [512, 512]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
NAME: Select
|
||||
KEYS: ['prompt']
|
||||
META_KEYS: ['image_size']
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -0,0 +1,345 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_textlora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$"
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.embedders.0.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionXL
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-xl-base-1.0@sd_xl_base_1.0.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.13025
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.0120
|
||||
USE_EMA: False
|
||||
LOAD_REFINER: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 320
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: [ 1, 2, 10 ]
|
||||
CONTEXT_DIM: 2048
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2816
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 1
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
USE_GRAD: True
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenCLIPEmbedder
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: hidden
|
||||
LAYER_IDX: 11
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "target_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
REFINER_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 384
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 4
|
||||
CONTEXT_DIM: [ 1280, 1280, 1280, 1280 ]
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2560
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
#
|
||||
REFINER_COND_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "aesthetic_score" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 5.0
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [1024, 1024]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bicubic
|
||||
SIZE: 1024
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCropXL
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'img', 'prompt', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
META_KEYS: [ 'data_key', 'img_path' ]
|
||||
- NAME: Rename
|
||||
INPUT_KEY: [ 'img', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
OUTPUT_KEY: [ 'image', 'original_size_as_tuple', 'target_size_as_tuple', 'crop_coords_top_left' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [ 1024, 1024 ]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
KEYS: [ 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
- NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
- NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
- NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
- NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
- NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -127,7 +127,7 @@ SOLVER:
|
||||
DOWN_RATIO: 1.0
|
||||
CONTROL_ANNO:
|
||||
NAME: HedAnnotator
|
||||
PRETRAINED_MODEL: ms://damo/scepter_scedit@annotator/ckpts/ControlNetHED.pth
|
||||
PRETRAINED_MODEL: ms://iic/scepter_scedit@annotator/ckpts/ControlNetHED.pth
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
|
||||
@@ -125,8 +125,8 @@ SOLVER:
|
||||
DOWN_RATIO: 1.0
|
||||
CONTROL_ANNO:
|
||||
NAME: OpenposeAnnotator
|
||||
BODY_MODEL_PATH: ms://damo/scepter_scedit@annotator/ckpts/body_pose_model.pth
|
||||
HAND_MODEL_PATH: ms://damo/scepter_scedit@annotator/ckpts/hand_pose_model.pth
|
||||
BODY_MODEL_PATH: ms://iic/scepter_scedit@annotator/ckpts/body_pose_model.pth
|
||||
HAND_MODEL_PATH: ms://iic/scepter_scedit@annotator/ckpts/hand_pose_model.pth
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
|
||||
@@ -241,7 +241,7 @@ SOLVER:
|
||||
DOWN_RATIO: 1.0
|
||||
CONTROL_ANNO:
|
||||
NAME: MidasDetector
|
||||
PRETRAINED_MODEL: ms://damo/scepter_scedit@annotator/ckpts/dpt_hybrid-midas-501f0c75.pt
|
||||
PRETRAINED_MODEL: ms://iic/scepter_scedit@annotator/ckpts/dpt_hybrid-midas-501f0c75.pt
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd15_512_textsce_t2i_swift
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftSCETuning
|
||||
DIMS: [1280, 1280, 1280, 1280, 1280, 640, 640, 640, 320, 320, 320, 320]
|
||||
TARGET_MODULES: model.lsc_identity\.\d+$
|
||||
DOWN_RATIO: 1.0
|
||||
TUNER_MODE: identity
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-v1-5@v1-5-pruned-emaonly.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
USE_GRAD: True
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 512
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 512
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [512, 512]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
NAME: Select
|
||||
KEYS: ['prompt']
|
||||
META_KEYS: ['image_size']
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -0,0 +1,348 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_textsce_t2i_swift
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftSCETuning
|
||||
DIMS: [1280, 1280, 640, 640, 640, 320, 320, 320, 320]
|
||||
TARGET_MODULES: model.lsc_identity\.\d+$
|
||||
DOWN_RATIO: 1.0
|
||||
TUNER_MODE: identity
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.embedders.0.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionXL
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-xl-base-1.0@sd_xl_base_1.0.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.13025
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.0120
|
||||
USE_EMA: False
|
||||
LOAD_REFINER: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 320
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: [ 1, 2, 10 ]
|
||||
CONTEXT_DIM: 2048
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2816
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 1
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
USE_GRAD: True
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenCLIPEmbedder
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: hidden
|
||||
LAYER_IDX: 11
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "target_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
REFINER_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 384
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 4
|
||||
CONTEXT_DIM: [ 1280, 1280, 1280, 1280 ]
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2560
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
REFINER_COND_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "aesthetic_score" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [1024, 1024]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bicubic
|
||||
SIZE: 1024
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCropXL
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'img', 'prompt', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
META_KEYS: [ 'data_key', 'img_path' ]
|
||||
- NAME: Rename
|
||||
INPUT_KEY: [ 'img', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
OUTPUT_KEY: [ 'image', 'original_size_as_tuple', 'target_size_as_tuple', 'crop_coords_top_left' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [ 1024, 1024 ]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
KEYS: [ 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -5,59 +5,59 @@ CONTROLLERS:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD2.1
|
||||
TYPE: Canny
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD2.1/canny_control/
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD2.1/canny_control/
|
||||
- NAME: openpose
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD2.1
|
||||
TYPE: Openpose
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD2.1/pose_control/
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD2.1/pose_control/
|
||||
- NAME: color
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD2.1
|
||||
TYPE: Color
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD2.1/color_control/
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD2.1/color_control/
|
||||
- NAME: hed
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD2.1
|
||||
TYPE: Hed
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD2.1/hed_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD2.1/hed_control
|
||||
- NAME: depth
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD2.1
|
||||
TYPE: Midas
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD2.1/depth_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD2.1/depth_control
|
||||
# SD_XL1.0
|
||||
- NAME: canny
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD_XL1.0
|
||||
TYPE: Canny
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD_XL1.0/canny_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD_XL1.0/canny_control
|
||||
- NAME: color
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD_XL1.0
|
||||
TYPE: Color
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD_XL1.0/color_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD_XL1.0/color_control
|
||||
- NAME: depth
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD_XL1.0
|
||||
TYPE: Midas
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD_XL1.0/depth_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD_XL1.0/depth_control
|
||||
- NAME: hed
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD_XL1.0
|
||||
TYPE: Hed
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD_XL1.0/hed_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD_XL1.0/hed_control
|
||||
- NAME: openpose
|
||||
NAME_ZH:
|
||||
DESCRIPTION:
|
||||
BASE_MODEL: SD_XL1.0
|
||||
TYPE: Openpose
|
||||
MODEL_PATH: ms://damo/scepter_scedit@controllable_model/SD_XL1.0/pose_control
|
||||
MODEL_PATH: ms://iic/scepter_scedit@controllable_model/SD_XL1.0/pose_control
|
||||
|
||||
@@ -1,74 +1,83 @@
|
||||
TUNERS:
|
||||
- NAME: Clay-Style-Editing
|
||||
NAME_ZH: 黏土风
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: EDIT
|
||||
MODEL_PATH: ms://iic/stylebooth@tuners/clay_style_edit/
|
||||
IMAGE_PATH: ms://iic/stylebooth@tuners/clay_style_edit/image.jpg
|
||||
TUNER_TYPE: LORA
|
||||
PROMPT_EXAMPLE: Convert this image into clay style
|
||||
- NAME: Azure-Dragon
|
||||
NAME_ZH: 青龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/azure_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/azure_dragon/xl_azure_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/azure_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/azure_dragon/xl_azure_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: Azure Dragon, 8K, high quality,Ultra High Detail.One of the Four Divine Creatures in Charge of Water.
|
||||
- NAME: Gold-Dragon
|
||||
NAME_ZH: 金龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/gold_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/gold_dragon/xl_gold_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/gold_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/gold_dragon/xl_gold_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: Chinese Gold Dragon in the clouds. Translucent Texture. Zbrush. Fuzzy Art. Exquisite Craftsmanship. 3D. 8K. Ultra High Detail
|
||||
- NAME: SpringFestival-Dragon
|
||||
NAME_ZH: 春节龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/spring_festival_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/spring_festival_dragon/xl_spring_festival_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/spring_festival_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/spring_festival_dragon/xl_spring_festival_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: Chinese dragon. Spring Festival.Festive.Street.Lanterns.32K.High quality.expressive, dramatic, dreamlike and mysterious, Surrealism
|
||||
- NAME: Red-Dragon
|
||||
NAME_ZH: 红龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/red_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/red_dragon/xl_red_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/red_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/red_dragon/xl_red_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: Traditional Red Dragon of China. Low Water Level. Studio Ghibli Style. Mural Illustration. White Background. High Detail
|
||||
- NAME: ChinesePunk-Dragon
|
||||
NAME_ZH: 中国朋克龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/chinese_punk_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/chinese_punk_dragon/xl_chinese_punk_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/chinese_punk_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/chinese_punk_dragon/xl_chinese_punk_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: uhd Image,Dragon,Chinese Dragon, Dunhuang Mural Style, Traditional Maritime Art Style
|
||||
- NAME: Cute-Dragon
|
||||
NAME_ZH: 喜庆龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/cute_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/cute_dragon/xl_kawaii_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/cute_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/cute_dragon/xl_kawaii_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: China Kawaii Dragon. Contest Winner. Minimalist Illustration. White Background. Flat Style. Digital Painting Style. Red. 32k uhd. Fun Comics. Fuzzy Art. Bold. Comic-Inspired Characters
|
||||
- NAME: Dragon-Baby
|
||||
NAME_ZH: 龙宝宝
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/baby_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/baby_dragon/xl_baby_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/baby_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/baby_dragon/xl_baby_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: Warm Colors, Soft,Chinese Dragon Baby, Felt Style,Dragon Baby, Best Quality, 3D Doll, Macaron Tones, Glittering Big Eyes, Winter,Dragon
|
||||
- NAME: Sloppy-Dragon
|
||||
NAME_ZH: 潦草龙
|
||||
SOURCE: wanx
|
||||
SOURCE: scepter
|
||||
DESCRIPTION: None
|
||||
BASE_MODEL: SD_XL1.0
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sloppy_dragon/
|
||||
IMAGE_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sloppy_dragon/xl_sloppy_dragon.png
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sloppy_dragon/
|
||||
IMAGE_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sloppy_dragon/xl_sloppy_dragon.png
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: Messy Chinese Dragon,Cute, Wu Guanzhong, Rough
|
||||
-
|
||||
@@ -77,8 +86,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/894f40ed44b37c3372e6a22b8ae577a4.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/Caricature
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/894f40ed44b37c3372e6a22b8ae577a4.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/Caricature
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -87,8 +96,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/894f40ed44b37c3372e6a22b8ae577a4.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/Caricature
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/894f40ed44b37c3372e6a22b8ae577a4.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/Caricature
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -97,8 +106,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/894f40ed44b37c3372e6a22b8ae577a4.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/Caricature
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/894f40ed44b37c3372e6a22b8ae577a4.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/Caricature
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -107,8 +116,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/80e5b4075c572c04cbb4e48c37b8366b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/ColorFieldPainting
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/80e5b4075c572c04cbb4e48c37b8366b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/ColorFieldPainting
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -117,8 +126,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/80e5b4075c572c04cbb4e48c37b8366b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/ColorFieldPainting
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/80e5b4075c572c04cbb4e48c37b8366b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/ColorFieldPainting
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -127,8 +136,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/80e5b4075c572c04cbb4e48c37b8366b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/ColorFieldPainting
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/80e5b4075c572c04cbb4e48c37b8366b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/ColorFieldPainting
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -137,8 +146,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/9ae235d7f1a7c2a4edab52a5e9f9cbae.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/ColoredPencilArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/9ae235d7f1a7c2a4edab52a5e9f9cbae.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/ColoredPencilArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -147,8 +156,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/9ae235d7f1a7c2a4edab52a5e9f9cbae.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/ColoredPencilArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/9ae235d7f1a7c2a4edab52a5e9f9cbae.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/ColoredPencilArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -157,8 +166,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/9ae235d7f1a7c2a4edab52a5e9f9cbae.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/ColoredPencilArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/9ae235d7f1a7c2a4edab52a5e9f9cbae.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/ColoredPencilArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -167,8 +176,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/3da915da2f5cedaf243e57e08163f35b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/DarkMoodyAtmosphere
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/3da915da2f5cedaf243e57e08163f35b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/DarkMoodyAtmosphere
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -177,8 +186,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/3da915da2f5cedaf243e57e08163f35b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/DarkMoodyAtmosphere
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/3da915da2f5cedaf243e57e08163f35b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/DarkMoodyAtmosphere
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -187,8 +196,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/3da915da2f5cedaf243e57e08163f35b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/DarkMoodyAtmosphere
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/3da915da2f5cedaf243e57e08163f35b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/DarkMoodyAtmosphere
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -197,8 +206,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/69fd81f5983107acc3d334af62915851.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/DrippingPaintSplatterArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/69fd81f5983107acc3d334af62915851.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/DrippingPaintSplatterArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -207,8 +216,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/69fd81f5983107acc3d334af62915851.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/DrippingPaintSplatterArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/69fd81f5983107acc3d334af62915851.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/DrippingPaintSplatterArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -217,8 +226,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/69fd81f5983107acc3d334af62915851.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/DrippingPaintSplatterArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/69fd81f5983107acc3d334af62915851.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/DrippingPaintSplatterArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -227,8 +236,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/f152edb4b3ca6248758b48115258ddfa.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/FadedPolaroidPhoto
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/f152edb4b3ca6248758b48115258ddfa.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/FadedPolaroidPhoto
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -237,8 +246,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/f152edb4b3ca6248758b48115258ddfa.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/FadedPolaroidPhoto
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/f152edb4b3ca6248758b48115258ddfa.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/FadedPolaroidPhoto
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -247,8 +256,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/f152edb4b3ca6248758b48115258ddfa.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/FadedPolaroidPhoto
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/f152edb4b3ca6248758b48115258ddfa.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/FadedPolaroidPhoto
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -257,8 +266,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/940cfd34155634cf051e1b2942cca426.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/Flat2DArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/940cfd34155634cf051e1b2942cca426.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/Flat2DArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -267,8 +276,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/940cfd34155634cf051e1b2942cca426.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/Flat2DArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/940cfd34155634cf051e1b2942cca426.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/Flat2DArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -277,8 +286,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/940cfd34155634cf051e1b2942cca426.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/Flat2DArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/940cfd34155634cf051e1b2942cca426.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/Flat2DArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -287,8 +296,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/57b751b11564cb22cd49ef21f2004a5f.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/GraffitiArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/57b751b11564cb22cd49ef21f2004a5f.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/GraffitiArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -297,8 +306,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/57b751b11564cb22cd49ef21f2004a5f.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/GraffitiArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/57b751b11564cb22cd49ef21f2004a5f.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/GraffitiArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -307,8 +316,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/57b751b11564cb22cd49ef21f2004a5f.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/GraffitiArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/57b751b11564cb22cd49ef21f2004a5f.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/GraffitiArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -317,8 +326,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/0312b673dc6858a9864d7f45f0c5c1fc.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/Impressionism
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/0312b673dc6858a9864d7f45f0c5c1fc.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/Impressionism
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -327,8 +336,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/0312b673dc6858a9864d7f45f0c5c1fc.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/Impressionism
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/0312b673dc6858a9864d7f45f0c5c1fc.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/Impressionism
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -337,8 +346,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/0312b673dc6858a9864d7f45f0c5c1fc.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/Impressionism
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/0312b673dc6858a9864d7f45f0c5c1fc.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/Impressionism
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -347,8 +356,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/9aa040b0c60d289da9610c91ad9b7c7e.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/LogoDesign
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/9aa040b0c60d289da9610c91ad9b7c7e.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/LogoDesign
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -357,8 +366,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/9aa040b0c60d289da9610c91ad9b7c7e.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/LogoDesign
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/9aa040b0c60d289da9610c91ad9b7c7e.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/LogoDesign
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -367,8 +376,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/9aa040b0c60d289da9610c91ad9b7c7e.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/LogoDesign
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/9aa040b0c60d289da9610c91ad9b7c7e.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/LogoDesign
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -377,8 +386,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/a9056e1eac85e5e4fe96a93917d4cce4.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/PencilSketchDrawing
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/a9056e1eac85e5e4fe96a93917d4cce4.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/PencilSketchDrawing
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -387,8 +396,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/a9056e1eac85e5e4fe96a93917d4cce4.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/PencilSketchDrawing
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/a9056e1eac85e5e4fe96a93917d4cce4.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/PencilSketchDrawing
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -397,8 +406,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/a9056e1eac85e5e4fe96a93917d4cce4.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/PencilSketchDrawing
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/a9056e1eac85e5e4fe96a93917d4cce4.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/PencilSketchDrawing
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -407,8 +416,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/568777f447fc02510b618152726d5002.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/SilhouetteArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/568777f447fc02510b618152726d5002.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/SilhouetteArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -417,8 +426,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/568777f447fc02510b618152726d5002.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/SilhouetteArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/568777f447fc02510b618152726d5002.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/SilhouetteArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -427,8 +436,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/568777f447fc02510b618152726d5002.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/SilhouetteArt
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/568777f447fc02510b618152726d5002.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/SilhouetteArt
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -437,8 +446,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/07d7b27cd73f2d43684003563511c15b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/Steampunk2
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/07d7b27cd73f2d43684003563511c15b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/Steampunk2
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -447,8 +456,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/07d7b27cd73f2d43684003563511c15b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/Steampunk2
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/07d7b27cd73f2d43684003563511c15b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/Steampunk2
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -457,8 +466,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/07d7b27cd73f2d43684003563511c15b.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/Steampunk2
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/07d7b27cd73f2d43684003563511c15b.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/Steampunk2
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -467,8 +476,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/2d1e9867058db2c57f2fe47530de3243.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/StickerDesigns
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/2d1e9867058db2c57f2fe47530de3243.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/StickerDesigns
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -477,8 +486,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/2d1e9867058db2c57f2fe47530de3243.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/StickerDesigns
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/2d1e9867058db2c57f2fe47530de3243.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/StickerDesigns
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -487,8 +496,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/2d1e9867058db2c57f2fe47530de3243.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/StickerDesigns
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/2d1e9867058db2c57f2fe47530de3243.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/StickerDesigns
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -497,8 +506,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/8859d532ae5901cc8457d6118fb9b7da.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/Watercolor2
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/8859d532ae5901cc8457d6118fb9b7da.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/Watercolor2
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -507,8 +516,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/8859d532ae5901cc8457d6118fb9b7da.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/Watercolor2
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/8859d532ae5901cc8457d6118fb9b7da.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/Watercolor2
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -517,8 +526,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: diva
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/8859d532ae5901cc8457d6118fb9b7da.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/Watercolor2
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/8859d532ae5901cc8457d6118fb9b7da.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/Watercolor2
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -527,8 +536,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/5895d78cf58c1ca05178991f37cc48ff.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/mre-elemental-art
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/5895d78cf58c1ca05178991f37cc48ff.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/mre-elemental-art
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -537,8 +546,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/5895d78cf58c1ca05178991f37cc48ff.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/mre-elemental-art
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/5895d78cf58c1ca05178991f37cc48ff.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/mre-elemental-art
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -547,8 +556,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/5895d78cf58c1ca05178991f37cc48ff.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/mre-elemental-art
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/5895d78cf58c1ca05178991f37cc48ff.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/mre-elemental-art
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -557,8 +566,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/a08149bc8e50f6bc65c0010d4cd416f8.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/mre-anime
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/a08149bc8e50f6bc65c0010d4cd416f8.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/mre-anime
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -567,8 +576,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/a08149bc8e50f6bc65c0010d4cd416f8.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/mre-anime
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/a08149bc8e50f6bc65c0010d4cd416f8.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/mre-anime
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -577,8 +586,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/a08149bc8e50f6bc65c0010d4cd416f8.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/mre-anime
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/a08149bc8e50f6bc65c0010d4cd416f8.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/mre-anime
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -587,8 +596,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/48c65cebf1fa4284d7b8feb619412e65.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/mre-comic
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/48c65cebf1fa4284d7b8feb619412e65.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/mre-comic
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -597,8 +606,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/48c65cebf1fa4284d7b8feb619412e65.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/mre-comic
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/48c65cebf1fa4284d7b8feb619412e65.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/mre-comic
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -607,8 +616,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: mre
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/48c65cebf1fa4284d7b8feb619412e65.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/mre-comic
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/48c65cebf1fa4284d7b8feb619412e65.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/mre-comic
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -617,8 +626,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/8fc51113f725f27326c4398a7457cd6d.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sai-craftclay
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/8fc51113f725f27326c4398a7457cd6d.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sai-craftclay
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -627,8 +636,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/8fc51113f725f27326c4398a7457cd6d.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/sai-craftclay
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/8fc51113f725f27326c4398a7457cd6d.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/sai-craftclay
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -637,8 +646,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/8fc51113f725f27326c4398a7457cd6d.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/sai-craftclay
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/8fc51113f725f27326c4398a7457cd6d.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/sai-craftclay
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -647,8 +656,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/a6f8d92afcd5803dfb2ebecbc92091b6.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sai-fantasyart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/a6f8d92afcd5803dfb2ebecbc92091b6.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sai-fantasyart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -657,8 +666,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/a6f8d92afcd5803dfb2ebecbc92091b6.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/sai-fantasyart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/a6f8d92afcd5803dfb2ebecbc92091b6.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/sai-fantasyart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -667,8 +676,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/a6f8d92afcd5803dfb2ebecbc92091b6.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/sai-fantasyart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/a6f8d92afcd5803dfb2ebecbc92091b6.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/sai-fantasyart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -677,8 +686,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/034a51b0dd34b018be8859bf45b4f7ed.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sai-lineart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/034a51b0dd34b018be8859bf45b4f7ed.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sai-lineart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -687,8 +696,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/034a51b0dd34b018be8859bf45b4f7ed.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/sai-lineart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/034a51b0dd34b018be8859bf45b4f7ed.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/sai-lineart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -697,8 +706,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/034a51b0dd34b018be8859bf45b4f7ed.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/sai-lineart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/034a51b0dd34b018be8859bf45b4f7ed.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/sai-lineart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -707,8 +716,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/7e9ed25bb34008beb5f417df63c4b2fe.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sai-neonpunk
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/7e9ed25bb34008beb5f417df63c4b2fe.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sai-neonpunk
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -717,8 +726,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/7e9ed25bb34008beb5f417df63c4b2fe.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/sai-neonpunk
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/7e9ed25bb34008beb5f417df63c4b2fe.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/sai-neonpunk
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -727,8 +736,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/7e9ed25bb34008beb5f417df63c4b2fe.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/sai-neonpunk
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/7e9ed25bb34008beb5f417df63c4b2fe.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/sai-neonpunk
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -737,8 +746,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/924f46a8f276011a0953d7988e90ee25.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sai-origami
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/924f46a8f276011a0953d7988e90ee25.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sai-origami
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -747,8 +756,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/924f46a8f276011a0953d7988e90ee25.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/sai-origami
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/924f46a8f276011a0953d7988e90ee25.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/sai-origami
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -757,8 +766,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/924f46a8f276011a0953d7988e90ee25.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/sai-origami
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/924f46a8f276011a0953d7988e90ee25.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/sai-origami
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -767,8 +776,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD_XL1.0
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD_XL1.0/a5ab89c0960be8c1216e65c98d92ae4a.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD_XL1.0/sai-pixelart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD_XL1.0/a5ab89c0960be8c1216e65c98d92ae4a.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD_XL1.0/sai-pixelart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -777,8 +786,8 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD2.1
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD2.1/a5ab89c0960be8c1216e65c98d92ae4a.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD2.1/sai-pixelart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD2.1/a5ab89c0960be8c1216e65c98d92ae4a.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD2.1/sai-pixelart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
-
|
||||
@@ -787,7 +796,7 @@ TUNERS:
|
||||
DESCRIPTION:
|
||||
SOURCE: sai
|
||||
BASE_MODEL: SD1.5
|
||||
IMAGE_PATH: ms://damo/scepter@mantra_images/SD1.5/a5ab89c0960be8c1216e65c98d92ae4a.png
|
||||
MODEL_PATH: ms://damo/scepter_scedit@tuners_model/SD1.5/sai-pixelart
|
||||
IMAGE_PATH: ms://iic/scepter@mantra_images_jpg/SD1.5/a5ab89c0960be8c1216e65c98d92ae4a.jpg
|
||||
MODEL_PATH: ms://iic/scepter_scedit@tuners_model/SD1.5/sai-pixelart
|
||||
TUNER_TYPE: SwiftSCE
|
||||
PROMPT_EXAMPLE: a boy wearing green jacket
|
||||
|
||||
@@ -10,7 +10,7 @@ DESC_INFO:
|
||||
<table align="center">
|
||||
<tr>
|
||||
<td>
|
||||
<img src="https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=assets/scepter_studio/scepter_studio_banner.jpg">
|
||||
<img src="https://modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets/scepter_studio/scepter_studio_banner.jpg">
|
||||
<h3><center>SCEPTER Studio是基于开源基模型和自研微调编辑算法构建的生成定制和编辑工具箱,提供围绕生成、微调、编辑、数据处理等一系列的工具和插件。</center><h3>
|
||||
</td>
|
||||
</tr>
|
||||
@@ -22,7 +22,7 @@ DESC_INFO:
|
||||
<table align="center">
|
||||
<tr>
|
||||
<td>
|
||||
<img src="https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=assets/scepter_studio/scepter_studio_banner.jpg">
|
||||
<img src="https://modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets/scepter_studio/scepter_studio_banner.jpg">
|
||||
<h3><center>SCEPTER Studio is a customized generation and editing toolkit built on the open-source base models and proprietary fine-tuning editing algorithms, offering a range of tools and plugins centered around generation, fine-tuning, editing, and data processing.</center><h3>
|
||||
</td>
|
||||
</tr>
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
NAME: EDIT
|
||||
IS_DEFAULT: False
|
||||
DEFAULT_PARAS:
|
||||
PARAS:
|
||||
RESOLUTIONS: [[512, 512], [1024, 1024]]
|
||||
INPUT:
|
||||
IMAGE:
|
||||
PROMPT: ""
|
||||
NEGATIVE_PROMPT: ""
|
||||
TARGET_SIZE_AS_TUPLE: [1024, 1024]
|
||||
PROMPT_PREFIX: ""
|
||||
SAMPLE: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
GUIDE_SCALE:
|
||||
text: 7.5
|
||||
image: 1.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
OUTPUT:
|
||||
LATENT:
|
||||
IMAGES:
|
||||
SEED:
|
||||
MODULES_PARAS:
|
||||
FIRST_STAGE_MODEL:
|
||||
FUNCTION:
|
||||
-
|
||||
NAME: encode
|
||||
DTYPE: float16
|
||||
INPUT: ["IMAGE"]
|
||||
-
|
||||
NAME: decode
|
||||
DTYPE: float16
|
||||
INPUT: ["LATENT"]
|
||||
PARAS:
|
||||
# SCALE_FACTOR DESCRIPTION: The vae embeding scale. TYPE: float default: 0.18215
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
DIFFUSION_MODEL:
|
||||
FUNCTION:
|
||||
-
|
||||
NAME: forward
|
||||
DTYPE: float16
|
||||
INPUT: ["SAMPLE_STEPS", "SAMPLE", "GUIDE_SCALE", "GUIDE_RESCALE", "DISCRETIZATION"]
|
||||
COND_STAGE_MODEL:
|
||||
FUNCTION:
|
||||
-
|
||||
NAME: encode_text
|
||||
DTYPE: float16
|
||||
INPUT: ["PROMPT", "NEGATIVE_PROMPT"]
|
||||
|
||||
MODEL:
|
||||
PRETRAINED_MODEL: ms://iic/stylebooth@models/stylebooth-tb-5000-0.bin
|
||||
SCHEDULE:
|
||||
PARAMETERIZATION: "eps"
|
||||
TIMESTEPS: 1000
|
||||
ZERO_TERMINAL_SNR: False
|
||||
SCHEDULE_ARGS:
|
||||
# NAME DESCRIPTION: TYPE: default: ''
|
||||
NAME: "scaled_linear"
|
||||
BETA_MIN: 0.00085
|
||||
BETA_MAX: 0.0120
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
PRETRAINED_PATH:
|
||||
IN_CHANNELS: 8
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
USE_GRAD: False
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
@@ -63,7 +63,8 @@ DIFFUSION_PARAS:
|
||||
MAX: 1.0
|
||||
DEFAULT: 0.15
|
||||
RESOLUTIONS:
|
||||
VALUES: [[704, 1408], [704, 1344], [768, 1344],
|
||||
VALUES: [ [512, 512], [768, 768],
|
||||
[704, 1408], [704, 1344], [768, 1344],
|
||||
[720, 1280],
|
||||
[768, 1280], [832, 1216], [832, 1152],
|
||||
[896, 1152], [896, 1088], [960, 1088],
|
||||
@@ -87,18 +88,18 @@ CONTROLABLE_ANNOTATORS:
|
||||
IS_DEFAULT: True
|
||||
-
|
||||
NAME: "HedAnnotator"
|
||||
PRETRAINED_MODEL: "ms://damo/scepter_scedit@annotator/ckpts/ControlNetHED.pth"
|
||||
PRETRAINED_MODEL: "ms://iic/scepter_scedit@annotator/ckpts/ControlNetHED.pth"
|
||||
TYPE: Hed
|
||||
IS_DEFAULT: False
|
||||
-
|
||||
NAME: "OpenposeAnnotator"
|
||||
BODY_MODEL_PATH: "ms://damo/scepter_scedit@annotator/ckpts/body_pose_model.pth"
|
||||
HAND_MODEL_PATH: "ms://damo/scepter_scedit@annotator/ckpts/hand_pose_model.pth"
|
||||
BODY_MODEL_PATH: "ms://iic/scepter_scedit@annotator/ckpts/body_pose_model.pth"
|
||||
HAND_MODEL_PATH: "ms://iic/scepter_scedit@annotator/ckpts/hand_pose_model.pth"
|
||||
TYPE: Openpose
|
||||
IS_DEFAULT: False
|
||||
-
|
||||
NAME: "MidasDetector"
|
||||
PRETRAINED_MODEL: "ms://damo/scepter_scedit@annotator/ckpts/dpt_hybrid-midas-501f0c75.pt"
|
||||
PRETRAINED_MODEL: "ms://iic/scepter_scedit@annotator/ckpts/dpt_hybrid-midas-501f0c75.pt"
|
||||
TYPE: Midas
|
||||
IS_DEFAULT: False
|
||||
-
|
||||
|
||||
@@ -68,7 +68,7 @@ DEFAULT_PARAS:
|
||||
INPUT: ["ORIGINAL_SIZE_AS_TUPLE", "AESTHETIC_SCORE", "NEGATIVE_AESTHETIC_SCORE", "CROP_COORDS_TOP_LEFT", "PROMPT", "NEGATIVE_PROMPT"]
|
||||
|
||||
MODEL:
|
||||
PRETRAINED_MODEL: ms://damo/LARGEN@models/largen_ckpt_s22k.pth
|
||||
PRETRAINED_MODEL: ms://iic/LARGEN@models/largen_ckpt_s22k.pth
|
||||
# SCHEDULE_ARGS DESCRIPTION: TYPE: default: ''
|
||||
SCHEDULE:
|
||||
PARAMETERIZATION: "eps"
|
||||
@@ -245,8 +245,8 @@ MODEL:
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: IPAdapterPlusEmbedder
|
||||
CLIP_DIR: ms://damo/LARGEN@models/clip_encoder/
|
||||
PRETRAINED_MODEL: ms://damo/LARGEN@models/ip-adapter-plus_sdxl_vit-h.bin
|
||||
CLIP_DIR: ms://iic/LARGEN@models/clip_encoder/
|
||||
PRETRAINED_MODEL: ms://iic/LARGEN@models/ip-adapter-plus_sdxl_vit-h.bin
|
||||
INPUT_KEYS: [ "ref_ip", "ref_detail" ]
|
||||
IN_DIM: 1280
|
||||
HEADS: 20
|
||||
|
||||
@@ -5,3 +5,169 @@ FILE_SYSTEM:
|
||||
# NAME DESCRIPTION: TYPE: default: ''
|
||||
NAME: LocalFs
|
||||
AUTO_CLEAN: False
|
||||
|
||||
PROCESSORS:
|
||||
- NAME: BlipImageBase
|
||||
TYPE: caption
|
||||
MODEL_PATH: ms://cubeai/blip-image-captioning-base
|
||||
DEVICE: "gpu"
|
||||
MEMORY: 1200
|
||||
PARAS:
|
||||
- LANGUAGE_NAME: English
|
||||
LANGUAGE_ZH_NAME: 英语
|
||||
- NAME: QWVL
|
||||
TYPE: caption
|
||||
MODEL_PATH: ms://qwen/Qwen-VL:v1.0.3
|
||||
DEVICE: "gpu"
|
||||
MEMORY: 19968
|
||||
PARAS:
|
||||
- PROMPT: 用中文描述这张图片
|
||||
LANGUAGE_NAME: Chinese
|
||||
LANGUAGE_ZH_NAME: 中文
|
||||
MAX_NEW_TOKENS:
|
||||
VALUE: 1024
|
||||
MAX: 2048
|
||||
STEP: 128
|
||||
MIN: 256
|
||||
MIN_NEW_TOKENS:
|
||||
VALUE: 16
|
||||
MAX: 1024
|
||||
STEP: 16
|
||||
MIN: 0
|
||||
NUM_BEAMS:
|
||||
VALUE: 1
|
||||
MAX: 12
|
||||
STEP: 1
|
||||
MIN: 1
|
||||
REPETITION_PENALTY:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
TEMPERATURE:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
- PROMPT: Generate the caption in English
|
||||
LANGUAGE_NAME: English
|
||||
LANGUAGE_ZH_NAME: 英语
|
||||
MAX_NEW_TOKENS:
|
||||
VALUE: 1024
|
||||
MAX: 2048
|
||||
STEP: 128
|
||||
MIN: 256
|
||||
MIN_NEW_TOKENS:
|
||||
VALUE: 16
|
||||
MAX: 1024
|
||||
STEP: 16
|
||||
MIN: 0
|
||||
NUM_BEAMS:
|
||||
VALUE: 1
|
||||
MAX: 12
|
||||
STEP: 1
|
||||
MIN: 1
|
||||
REPETITION_PENALTY:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
TEMPERATURE:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
-
|
||||
NAME: QWVLQuantize
|
||||
TYPE: caption
|
||||
DEVICE: "gpu"
|
||||
MEMORY: 7885
|
||||
MODEL_PATH: ms://qwen/Qwen-VL:v1.0.3
|
||||
PARAS:
|
||||
- PROMPT: 用中文描述这张图片
|
||||
LANGUAGE_NAME: Chinese
|
||||
LANGUAGE_ZH_NAME: 中文
|
||||
MAX_NEW_TOKENS:
|
||||
VALUE: 1024
|
||||
MAX: 2048
|
||||
STEP: 128
|
||||
MIN: 256
|
||||
MIN_NEW_TOKENS:
|
||||
VALUE: 16
|
||||
MAX: 1024
|
||||
STEP: 16
|
||||
MIN: 0
|
||||
NUM_BEAMS:
|
||||
VALUE: 1
|
||||
MAX: 12
|
||||
STEP: 1
|
||||
MIN: 1
|
||||
REPETITION_PENALTY:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
TEMPERATURE:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
- PROMPT: Generate the caption in English
|
||||
LANGUAGE_NAME: English
|
||||
LANGUAGE_ZH_NAME: 英语
|
||||
MAX_NEW_TOKENS:
|
||||
VALUE: 1024
|
||||
MAX: 2048
|
||||
STEP: 128
|
||||
MIN: 256
|
||||
MIN_NEW_TOKENS:
|
||||
VALUE: 16
|
||||
MAX: 1024
|
||||
STEP: 16
|
||||
MIN: 0
|
||||
NUM_BEAMS:
|
||||
VALUE: 1
|
||||
MAX: 12
|
||||
STEP: 1
|
||||
MIN: 1
|
||||
REPETITION_PENALTY:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
TEMPERATURE:
|
||||
VALUE: 1.0
|
||||
MAX: 100.0
|
||||
STEP: 1.0
|
||||
MIN: 1.0
|
||||
-
|
||||
NAME: CenterCrop
|
||||
TYPE: image
|
||||
DEVICE: "cpu"
|
||||
MEMORY: 10
|
||||
PARAS:
|
||||
HEIGHT_RATIO:
|
||||
VALUE: 1
|
||||
MAX: 20
|
||||
STEP: 1
|
||||
MIN: 1
|
||||
WIDTH_RATIO:
|
||||
VALUE: 1
|
||||
MAX: 20
|
||||
STEP: 1
|
||||
MIN: 1
|
||||
# - NAME: PaddingCrop
|
||||
# TYPE: image
|
||||
# DEVICE: "cpu"
|
||||
# MEMORY: 10
|
||||
# PARAS:
|
||||
# HEIGHT_RATIO:
|
||||
# VALUE: 3
|
||||
# MAX: 25
|
||||
# STEP: 1
|
||||
# MIN: 1
|
||||
# WIDTH_RATIO:
|
||||
# VALUE: 4
|
||||
# MAX: 20
|
||||
# STEP: 1
|
||||
# MIN: 1
|
||||
|
||||
@@ -48,11 +48,11 @@ BANNER: |
|
||||
</div>
|
||||
<div class="qr-codes">
|
||||
<div class="qr-code-container">
|
||||
<img src="https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=assets/scepter_studio/ms_scepter_studio_qr.png" alt="ms_scepter_studio_qr">
|
||||
<img src="https://modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets/scepter_studio/ms_scepter_studio_qr.png" alt="ms_scepter_studio_qr">
|
||||
<div class="caption"><a href="https://www.modelscope.cn/studios/iic/scepter_studio">Modelscope Studio</a></div>
|
||||
</div>
|
||||
<div class="qr-code-container">
|
||||
<img src="https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=assets/scepter_studio/scepter_github_qr.png" alt="scepter_github_qr">
|
||||
<img src="https://modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets/scepter_studio/scepter_github_qr.png" alt="scepter_github_qr">
|
||||
<div class="caption"><a href="https://github.com/modelscope/scepter">Github</a></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -0,0 +1,376 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
META:
|
||||
VERSION: 'EDIT'
|
||||
DESCRIPTION: "EDIT"
|
||||
IS_DEFAULT: False
|
||||
IS_SHARE: True
|
||||
INFERENCE_PARAS:
|
||||
INFERENCE_BATCH_SIZE: 1
|
||||
INFERENCE_PREFIX: ""
|
||||
DEFAULT_SAMPLER: "ddim"
|
||||
DEFAULT_SAMPLE_STEPS: 40
|
||||
INFERENCE_N_PROMPT: ""
|
||||
RESOLUTION: [512, 512]
|
||||
PARAS:
|
||||
-
|
||||
TASK: 'Image Editing'
|
||||
TRAIN_BATCH_SIZE: 1
|
||||
TRAIN_PREFIX: ""
|
||||
TRAIN_N_PROMPT: ""
|
||||
RESOLUTION: [512, 512]
|
||||
MEMORY: 29000
|
||||
EPOCHS: 50
|
||||
SAVE_INTERVAL: 25
|
||||
EPSEC: 0.818
|
||||
LEARNING_RATE: 0.0001
|
||||
IS_DEFAULT: False
|
||||
TUNER: FULL
|
||||
-
|
||||
TASK: 'Image Editing'
|
||||
TRAIN_BATCH_SIZE: 1
|
||||
TRAIN_PREFIX: ""
|
||||
TRAIN_N_PROMPT: ""
|
||||
RESOLUTION: [512, 512]
|
||||
MEMORY: 29000
|
||||
EPOCHS: 50
|
||||
SAVE_INTERVAL: 25
|
||||
EPSEC: 0.818
|
||||
LEARNING_RATE: 0.0001
|
||||
IS_DEFAULT: True
|
||||
TUNER: LORA
|
||||
-
|
||||
TASK: 'Image Editing'
|
||||
TRAIN_BATCH_SIZE: 1
|
||||
TRAIN_PREFIX: ""
|
||||
TRAIN_N_PROMPT: ""
|
||||
RESOLUTION: [512, 512]
|
||||
MEMORY: 29000
|
||||
EPOCHS: 50
|
||||
SAVE_INTERVAL: 25
|
||||
EPSEC: 0.818
|
||||
LEARNING_RATE: 0.0001
|
||||
IS_DEFAULT: False
|
||||
TUNER: SCE
|
||||
|
||||
-
|
||||
TASK: 'Image Editing'
|
||||
TRAIN_BATCH_SIZE: 1
|
||||
TRAIN_PREFIX: ""
|
||||
TRAIN_N_PROMPT: ""
|
||||
RESOLUTION: [512, 512]
|
||||
MEMORY: 29000
|
||||
EPOCHS: 50
|
||||
SAVE_INTERVAL: 25
|
||||
EPSEC: 0.818
|
||||
LEARNING_RATE: 0.0001
|
||||
IS_DEFAULT: False
|
||||
TUNER: TEXT_SCE
|
||||
|
||||
-
|
||||
TASK: 'Image Editing'
|
||||
TRAIN_BATCH_SIZE: 4
|
||||
TRAIN_PREFIX: ""
|
||||
TRAIN_N_PROMPT: ""
|
||||
RESOLUTION: [512, 512]
|
||||
MEMORY: 29000
|
||||
EPOCHS: 50
|
||||
SAVE_INTERVAL: 25
|
||||
EPSEC: 0.818
|
||||
LEARNING_RATE: 0.0001
|
||||
IS_DEFAULT: False
|
||||
TUNER: TEXT_LORA
|
||||
|
||||
TUNERS:
|
||||
LORA:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 256
|
||||
LORA_ALPHA: 256
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$"
|
||||
TEXT_LORA:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 256
|
||||
LORA_ALPHA: 256
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "(cond_stage_model.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2))|(model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2))$"
|
||||
SCE:
|
||||
-
|
||||
NAME: SwiftSCETuning
|
||||
DIMS: [1280, 1280, 1280, 1280, 1280, 640, 640, 640, 320, 320, 320, 320]
|
||||
DOWN_RATIO: 1.0
|
||||
TARGET_MODULES: model.lsc_identity\.\d+$
|
||||
TUNER_MODE: identity
|
||||
TEXT_SCE:
|
||||
-
|
||||
NAME: SwiftSCETuning
|
||||
DIMS: [ 1280, 1280, 1280, 1280, 1280, 640, 640, 640, 320, 320, 320, 320 ]
|
||||
DOWN_RATIO: 1.0
|
||||
TARGET_MODULES: model.lsc_identity\.\d+$
|
||||
TUNER_MODE: identity
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 256
|
||||
LORA_ALPHA: 256
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
|
||||
MODIFY_PARAS:
|
||||
TEXT_LORA:
|
||||
TRAIN:
|
||||
SOLVER.MODEL.COND_STAGE_MODEL.USE_GRAD: True
|
||||
TEXT_SCE:
|
||||
TRAIN:
|
||||
SOLVER.MODEL.COND_STAGE_MODEL.USE_GRAD: True
|
||||
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 1000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: -1
|
||||
#
|
||||
WORK_DIR:
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
#
|
||||
FREEZE:
|
||||
#
|
||||
TUNER:
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionEdit
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://iic/stylebooth@models/stylebooth-tb-5000-0.bin
|
||||
IGNORE_KEYS: [ ]
|
||||
CONCAT_NO_SCALE_FACTOR: True
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 8
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
USE_GRAD: False
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: #7.5
|
||||
image: 1.5
|
||||
text: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: cache/save_data/delogo/
|
||||
MS_DATASET_NAMESPACE: ""
|
||||
MS_DATASET_SPLIT: "train"
|
||||
MS_DATASET_SUBNAME: ""
|
||||
PROMPT_PREFIX: ""
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFileList
|
||||
FILE_KEYS: ['img_path', 'src_path']
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'image', 'condition_cat' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'condition_cat', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 10
|
||||
SHOW_GPU_MEM: True
|
||||
-
|
||||
NAME: TensorboardLogHook
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 10000
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: Text2ImageDataset
|
||||
MODE: eval
|
||||
PROMPT_FILE:
|
||||
PROMPT_DATA: [ ]
|
||||
IMAGE_SIZE: [ 512, 512 ]
|
||||
FIELDS: [ "prompt", "src_path" ]
|
||||
DELIMITER: '#;#'
|
||||
PROMPT_PREFIX: ''
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFileList
|
||||
FILE_KEYS: [ 'src_path' ]
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'condition_cat' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'condition_cat', 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
SAVE_PROBE_PREFIX: 'image'
|
||||
@@ -5,6 +5,7 @@ META:
|
||||
VERSION: 'SD_XL1.0'
|
||||
DESCRIPTION: "Stable Diffusion XL1.0"
|
||||
IS_DEFAULT: True
|
||||
IS_SHARE: True
|
||||
INFERENCE_PARAS:
|
||||
INFERENCE_BATCH_SIZE: 1
|
||||
INFERENCE_PREFIX: ""
|
||||
@@ -532,7 +533,6 @@ SOLVER:
|
||||
GUIDE_SCALE: 5.0
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: linspace
|
||||
IMAGE_SIZE: [ 1024, 1024]
|
||||
RUN_TRAIN_N: False
|
||||
# OPTIMIZER DESCRIPTION: TYPE: default: ''
|
||||
OPTIMIZER:
|
||||
@@ -621,6 +621,7 @@ SOLVER:
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
|
||||
@@ -4,6 +4,7 @@ META:
|
||||
VERSION: 'SD1.5'
|
||||
DESCRIPTION: "Stable Diffusion v1.5"
|
||||
IS_DEFAULT: False
|
||||
IS_SHARE: True
|
||||
INFERENCE_PARAS:
|
||||
INFERENCE_BATCH_SIZE: 1
|
||||
INFERENCE_PREFIX: ""
|
||||
@@ -244,7 +245,6 @@ SOLVER:
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
@@ -274,14 +274,14 @@ SOLVER:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 512
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 512
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
@@ -332,6 +332,7 @@ SOLVER:
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
|
||||
@@ -4,6 +4,7 @@ META:
|
||||
VERSION: 'SD2.1'
|
||||
DESCRIPTION: "Stable Diffusion v2.1"
|
||||
IS_DEFAULT: False
|
||||
IS_SHARE: True
|
||||
INFERENCE_PARAS:
|
||||
INFERENCE_BATCH_SIZE: 1
|
||||
INFERENCE_PREFIX: ""
|
||||
@@ -186,7 +187,6 @@ SOLVER:
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [768, 768]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
@@ -216,14 +216,14 @@ SOLVER:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 768
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 768, 768 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 768
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 768, 768 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
@@ -274,6 +274,7 @@ SOLVER:
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
---
|
||||
frameworks:
|
||||
- Pytorch
|
||||
license: apache-2.0
|
||||
tasks:
|
||||
- efficient-diffusion-tuning
|
||||
---
|
||||
|
||||
<p align="center">
|
||||
|
||||
<h2 align="center">{MODEL_NAME}</h2>
|
||||
<p align="center">
|
||||
<br>
|
||||
<a href="https://github.com/modelscope/scepter/"><img src="https://img.shields.io/badge/powered by-scepter-6FEBB9.svg"></a>
|
||||
<br>
|
||||
</p>
|
||||
|
||||
## Model Introduction
|
||||
{MODEL_DESCRIPTION}
|
||||
|
||||
## Model Parameters
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th rowspan="2">Base Model</th>
|
||||
<th rowspan="2">Tuner Type</th>
|
||||
<th colspan="4">Training Parameters</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<th>Batch Size</th>
|
||||
<th>Epochs</th>
|
||||
<th>Learning Rate</th>
|
||||
<th>Resolution</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">{BASE_MODEL}</td>
|
||||
<td>{TUNER_TYPE}</td>
|
||||
<td>{TRAIN_BATCH_SIZE}</td>
|
||||
<td>{TRAIN_EPOCH}</td>
|
||||
<td>{LEARNING_RATE}</td>
|
||||
<td>[{HEIGHT}, {WIDTH}]</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Data Type</th>
|
||||
<th>Data Space</th>
|
||||
<th>Data Name</th>
|
||||
<th>Data Subset</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td>{DATA_TYPE}</td>
|
||||
<td>{MS_DATA_SPACE}</td>
|
||||
<td>{MS_DATA_NAME}</td>
|
||||
<td>{MS_DATA_SUBNAME}</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
|
||||
## Model Performance
|
||||
Given the input "{EVAL_PROMPT}," the following image may be generated:
|
||||
|
||||

|
||||
|
||||
## Model Usage
|
||||
### Command Line Execution
|
||||
* Run using Scepter's SDK, taking care to use different configuration files in accordance with the different base models, as per the corresponding relationships shown below
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th rowspan="2">Base Model</th>
|
||||
<th rowspan="1">LORA</th>
|
||||
<th colspan="1">SCE</th>
|
||||
<th colspan="1">TEXT_LORA</th>
|
||||
<th colspan="1">TEXT_SCE</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">SD1.5</td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_1.5_512_lora.yaml">lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sd15_512_sce_t2i_swift.yaml">sce_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_1.5_512_text_lora.yaml">text_lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/stable_diffusion_1.5_512_text_sce.yaml">text_sce_cfg</a></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">SD2.1</td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_2.1_768_lora.yaml">lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sd21_768_sce_t2i_swift.yaml">sce_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_2.1_768_text_lora.yaml">text_lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sd21_768_text_sce_t2i_swift.yaml">text_sce_cfg</a></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">SDXL</td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_xl_1024_lora.yaml">lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sdxl_1024_sce_t2i_swift.yaml">sce_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_xl_1024_text_lora.yaml">text_lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sdxl_1024_text_sce_t2i_swift.yaml">text_sce_cfg</a></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
* Running from Source Code
|
||||
|
||||
```shell
|
||||
git clone https://github.com/modelscope/scepter.git
|
||||
cd scepter
|
||||
pip install -r requirements/recommended.txt
|
||||
PYTHONPATH=. python scepter/tools/run_inference.py
|
||||
--pretrained_model {this model folder}
|
||||
--cfg {lora_cfg} or {sce_cfg} or {text_lora_cfg} or {text_sce_cfg}
|
||||
--prompt '{EVAL_PROMPT}'
|
||||
--save_folder 'inference'
|
||||
```
|
||||
|
||||
* Running after Installing Scepter (Recommended)
|
||||
```shell
|
||||
pip install scepter
|
||||
python -m scepter/tools/run_inference.py
|
||||
--pretrained_model {this model folder}
|
||||
--cfg {lora_cfg} or {sce_cfg} or {text_lora_cfg} or {text_sce_cfg}
|
||||
--prompt '{EVAL_PROMPT}'
|
||||
--save_folder 'inference'
|
||||
```
|
||||
### Running with Scepter Studio
|
||||
|
||||
```shell
|
||||
pip install scepter
|
||||
# Launch Scepter Studio
|
||||
python -m scepter.tools.webui
|
||||
```
|
||||
|
||||
* Refer to the following guides for model usage.
|
||||
|
||||
(video url)
|
||||
|
||||
## Model Reference
|
||||
If you wish to use this model for your own purposes, please cite it as follows.
|
||||
```bibtex
|
||||
@misc{{MODEL_NAME},
|
||||
title = {{MODEL_NAME}, {MODEL_URL}},
|
||||
author = {{USER_NAME}},
|
||||
year = {2024}
|
||||
}
|
||||
```
|
||||
This model was trained using [Scepter Studio](https://github.com/modelscope/scepter); [Scepter](https://github.com/modelscope/scepter)
|
||||
is an algorithm framework and toolbox developed by the Alibaba Tongyi Wanxiang Team. It provides a suite of tools and models for image generation, editing, fine-tuning, data processing, and more. If you find our work beneficial for your research,
|
||||
please cite as follows.
|
||||
```bibtex
|
||||
@misc{scepter,
|
||||
title = {SCEPTER, https://github.com/modelscope/scepter},
|
||||
author = {SCEPTER},
|
||||
year = {2023}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,168 @@
|
||||
---
|
||||
frameworks:
|
||||
- Pytorch
|
||||
license: apache-2.0
|
||||
tasks:
|
||||
- efficient-diffusion-tuning
|
||||
---
|
||||
|
||||
<p align="center">
|
||||
|
||||
<h2 align="center">{MODEL_NAME}</h2>
|
||||
<p align="center">
|
||||
<br>
|
||||
<a href="https://github.com/modelscope/scepter/"><img src="https://img.shields.io/badge/powered by-scepter-6FEBB9.svg"></a>
|
||||
<br>
|
||||
</p>
|
||||
|
||||
## 模型介绍
|
||||
{MODEL_DESCRIPTION}
|
||||
|
||||
## 模型参数
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th rowspan="2">基础模型</th>
|
||||
<th rowspan="2">微调类型</th>
|
||||
<th colspan="4">训练参数</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<th>批次大小</th>
|
||||
<th>轮数</th>
|
||||
<th>学习率</th>
|
||||
<th>分辨率</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">{BASE_MODEL}</td>
|
||||
<td>{TUNER_TYPE}</td>
|
||||
<td>{TRAIN_BATCH_SIZE}</td>
|
||||
<td>{TRAIN_EPOCH}</td>
|
||||
<td>{LEARNING_RATE}</td>
|
||||
<td>[{HEIGHT}, {WIDTH}]</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>数据类型</th>
|
||||
<th>数据空间</th>
|
||||
<th>数据名称</th>
|
||||
<th>数据子集</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td> {DATA_TYPE}</td>
|
||||
<td>{MS_DATA_SPACE}</td>
|
||||
<td>{MS_DATA_NAME}</td>
|
||||
<td>{MS_DATA_SUBNAME}</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
|
||||
## 模型效果
|
||||
|
||||
输入 "{EVAL_PROMPT}",可能会得到如下图像:
|
||||
|
||||

|
||||
|
||||
|
||||
## 模型使用
|
||||
### 命令行运行
|
||||
|
||||
* 使用scepter的sdk进行运行,注意需要按照模型参数中基模型的不同使用不同的配置文件,其对应关系如下
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th rowspan="2">Base Model</th>
|
||||
<th rowspan="1">LORA</th>
|
||||
<th colspan="1">SCE</th>
|
||||
<th colspan="1">TEXT_LORA</th>
|
||||
<th colspan="1">TEXT_SCE</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">SD1.5</td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_1.5_512_lora.yaml">lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sd15_512_sce_t2i_swift.yaml">sce_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_1.5_512_text_lora.yaml">text_lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/stable_diffusion_1.5_512_text_sce.yaml">text_sce_cfg</a></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">SD2.1</td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_2.1_768_lora.yaml">lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sd21_768_sce_t2i_swift.yaml">sce_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_2.1_768_text_lora.yaml">text_lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sd21_768_text_sce_t2i_swift.yaml">text_sce_cfg</a></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
<tbody align="center">
|
||||
<tr>
|
||||
<td rowspan="8">SDXL</td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_xl_1024_lora.yaml">lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sdxl_1024_sce_t2i_swift.yaml">sce_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/examples/generation/stable_diffusion_xl_1024_text_lora.yaml">text_lora_cfg</a></td>
|
||||
<td><a href="https://github.com/modelscope/scepter/blob/main/scepter/methods/scedit/t2i/sdxl_1024_text_sce_t2i_swift.yaml">text_sce_cfg</a></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
* 从源码运行
|
||||
|
||||
```shell
|
||||
git clone https://github.com/modelscope/scepter.git
|
||||
cd scepter
|
||||
pip install -r requirements/recommended.txt
|
||||
PYTHONPATH=. python scepter/tools/run_inference.py
|
||||
--pretrained_model {this model folder}
|
||||
--cfg {lora_cfg} or {sce_cfg} or {text_lora_cfg} or {text_sce_cfg}
|
||||
--prompt '{EVAL_PROMPT}'
|
||||
--save_folder 'inference'
|
||||
```
|
||||
|
||||
* 安装scepter后运行(推荐)
|
||||
```shell
|
||||
pip install scepter
|
||||
python -m scepter/tools/run_inference.py
|
||||
--pretrained_model {this model folder}
|
||||
--cfg {lora_cfg} or {sce_cfg} or {text_lora_cfg} or {text_sce_cfg}
|
||||
--prompt '{EVAL_PROMPT}'
|
||||
--save_folder 'inference'
|
||||
```
|
||||
### 使用Scepter Studio运行
|
||||
```shell
|
||||
pip install scepter
|
||||
启动scepter studio
|
||||
python -m scepter.tools.webui
|
||||
```
|
||||
* 参考以下指南使用模型
|
||||
|
||||
|
||||
## 模型引用
|
||||
如果你想使用该模型应用于自己的场景,请按照如下方式引用该模型。
|
||||
```bibtex
|
||||
@misc{{MODEL_NAME},
|
||||
title = {{MODEL_NAME}, {MODEL_URL}},
|
||||
author = {{USER_NAME}},
|
||||
year = {2024}
|
||||
}
|
||||
```
|
||||
该模型是基于[Scepter Studio](https://github.com/modelscope/scepter)训练得到;[scepter](https://github.com/modelscope/scepter)
|
||||
是由阿里巴巴通义万相团队开发的算法框架和工具箱,提供图像生成、编辑、微调、数据处理等一系列工具和模型。如果您觉得我们的工作有益于您的工作,
|
||||
请按照如下方式引用。
|
||||
```bibtex
|
||||
@misc{scepter,
|
||||
title = {SCEPTER, https://github.com/modelscope/scepter},
|
||||
author = {SCEPTER},
|
||||
year = {2023}
|
||||
}
|
||||
```
|
||||
@@ -1,2 +1,15 @@
|
||||
WORK_DIR: "tuner_manager"
|
||||
EXPORT_DIR: "export_model"
|
||||
TUNER_LIST_YAML: "tuner_list.yaml"
|
||||
README_EN: "scepter/methods/studio/tuner_manager/readme_en.md"
|
||||
README_ZH: "scepter/methods/studio/tuner_manager/readme_zh.md"
|
||||
|
||||
BASE_MODEL_VERSION:
|
||||
- BASE_MODEL: 'SD1.5'
|
||||
TUNER_TYPE: [ 'TEXT_LORA', 'TEXT_SCE', 'LORA', 'SCE', 'FULL' ]
|
||||
- BASE_MODEL: 'SD_XL1.0'
|
||||
TUNER_TYPE: [ 'TEXT_LORA', 'TEXT_SCE', 'LORA', 'SCE', 'FULL' ]
|
||||
- BASE_MODEL: 'SD2.1'
|
||||
TUNER_TYPE: [ 'LORA', 'SCE', 'FULL' ]
|
||||
- BASE_MODEL: 'EDIT'
|
||||
TUNER_TYPE: [ 'LORA', 'SCE', 'FULL' ]
|
||||
|
||||
@@ -4,7 +4,6 @@ from abc import ABCMeta
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
from scepter.modules.annotator.registry import ANNOTATORS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
|
||||
@@ -6,7 +6,6 @@ import cv2
|
||||
import numpy as np
|
||||
import torch
|
||||
from PIL import Image
|
||||
|
||||
from scepter.modules.annotator.base_annotator import BaseAnnotator
|
||||
from scepter.modules.annotator.registry import ANNOTATORS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
|
||||
@@ -6,7 +6,6 @@ import cv2
|
||||
import numpy as np
|
||||
import torch
|
||||
from PIL import Image
|
||||
|
||||
from scepter.modules.annotator.base_annotator import BaseAnnotator
|
||||
from scepter.modules.annotator.registry import ANNOTATORS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
|
||||
@@ -11,13 +11,14 @@ from abc import ABCMeta
|
||||
import cv2
|
||||
import numpy as np
|
||||
import torch
|
||||
import torchvision
|
||||
from einops import rearrange
|
||||
|
||||
from scepter.modules.annotator.base_annotator import BaseAnnotator
|
||||
from scepter.modules.annotator.registry import ANNOTATORS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
from scepter.modules.utils.distribute import we
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from torchvision.transforms import InterpolationMode
|
||||
|
||||
|
||||
def nms(x, t, s):
|
||||
@@ -123,6 +124,8 @@ class HedAnnotator(BaseAnnotator, metaclass=ABCMeta):
|
||||
if len(image.shape) == 3:
|
||||
image = rearrange(image, 'h w c -> 1 c h w')
|
||||
B, C, H, W = image.shape
|
||||
elif len(image.shape) == 4:
|
||||
B, C, H, W = image.shape
|
||||
else:
|
||||
raise "Unsurpport input image's shape"
|
||||
elif isinstance(image, np.ndarray):
|
||||
@@ -130,22 +133,22 @@ class HedAnnotator(BaseAnnotator, metaclass=ABCMeta):
|
||||
if len(image.shape) == 3:
|
||||
image = rearrange(image, 'h w c -> 1 c h w')
|
||||
B, C, H, W = image.shape
|
||||
elif len(image.shape) == 4:
|
||||
B, C, H, W = image.shape
|
||||
else:
|
||||
raise "Unsurpport input image's shape"
|
||||
else:
|
||||
raise "Unsurpport input image's type"
|
||||
transform = torchvision.transforms.Resize(
|
||||
(H, W), interpolation=InterpolationMode.BILINEAR, antialias=True)
|
||||
edges = self.netNetwork(image.to(we.device_id))
|
||||
edges = [
|
||||
e.detach().cpu().numpy().astype(np.float32)[0, 0] for e in edges
|
||||
]
|
||||
edges = [
|
||||
cv2.resize(e, (W, H), interpolation=cv2.INTER_LINEAR)
|
||||
for e in edges
|
||||
]
|
||||
edges = np.stack(edges, axis=2)
|
||||
edge = 1 / (1 + np.exp(-np.mean(edges, axis=2).astype(np.float64)))
|
||||
edge = 255 - (edge * 255.0).clip(0, 255).astype(np.uint8)
|
||||
return edge[..., None].repeat(3, 2)
|
||||
edges = [transform(e) for e in edges]
|
||||
edges = torch.cat(edges, dim=1)
|
||||
edges = 1 / (1 +
|
||||
torch.exp(-torch.mean(edges, dim=1).type(torch.float)))
|
||||
edges = edges.cpu().numpy()
|
||||
edges = 255 - (edges * 255.0).clip(0, 255).astype(np.uint8)
|
||||
return edges[..., None].repeat(3, -1)
|
||||
|
||||
@staticmethod
|
||||
def get_config_template():
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
# based on https://github.com/isl-org/MiDaS
|
||||
|
||||
import cv2
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import torch
|
||||
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
"""MidashNet: Network for monocular depth estimation trained by mixing several datasets.
|
||||
This file contains code that is adapted from
|
||||
https://github.com/thomasjpfan/pytorch_refinenet/blob/master/pytorch_refinenet/refinenet/refinenet_4cascade.py
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
"""MidashNet: Network for monocular depth estimation trained by mixing several datasets.
|
||||
This file contains code that is adapted from
|
||||
https://github.com/thomasjpfan/pytorch_refinenet/blob/master/pytorch_refinenet/refinenet/refinenet_4cascade.py
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import math
|
||||
|
||||
import cv2
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
"""Utils for monoDepth."""
|
||||
import re
|
||||
import sys
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import math
|
||||
import types
|
||||
|
||||
|
||||
@@ -9,7 +9,6 @@ import numpy as np
|
||||
import torch
|
||||
from einops import rearrange
|
||||
from PIL import Image
|
||||
|
||||
from scepter.modules.annotator.base_annotator import BaseAnnotator
|
||||
from scepter.modules.annotator.midas.api import MiDaSInference
|
||||
from scepter.modules.annotator.registry import ANNOTATORS
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.utils.model_zoo as model_zoo
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
# modified by lihaoweicv
|
||||
# pytorch version
|
||||
|
||||
@@ -11,7 +11,6 @@ import cv2
|
||||
import numpy as np
|
||||
import torch
|
||||
from PIL import Image
|
||||
|
||||
from scepter.modules.annotator.base_annotator import BaseAnnotator
|
||||
from scepter.modules.annotator.mlsd.mbv2_mlsd_large import MobileV2_MLSD_Large
|
||||
from scepter.modules.annotator.mlsd.utils import pred_lines
|
||||
|
||||
@@ -15,13 +15,12 @@ import numpy as np
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
from PIL import Image
|
||||
from scipy.ndimage.filters import gaussian_filter
|
||||
from skimage.measure import label
|
||||
|
||||
from scepter.modules.annotator.base_annotator import BaseAnnotator
|
||||
from scepter.modules.annotator.registry import ANNOTATORS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scipy.ndimage.filters import gaussian_filter
|
||||
from skimage.measure import label
|
||||
|
||||
os.environ['KMP_DUPLICATE_LIB_OK'] = 'TRUE'
|
||||
|
||||
|
||||
@@ -37,30 +37,30 @@ class AnnotatorProcessor():
|
||||
hed_cfg = {
|
||||
'NAME': 'HedAnnotator',
|
||||
'PRETRAINED_MODEL':
|
||||
'ms://damo/scepter_scedit@annotator/ckpts/ControlNetHED.pth',
|
||||
'ms://iic/scepter_scedit@annotator/ckpts/ControlNetHED.pth',
|
||||
'INPUT_KEYS': ['img'],
|
||||
'OUTPUT_KEYS': ['hed']
|
||||
}
|
||||
openpose_cfg = {
|
||||
'NAME': 'OpenposeAnnotator',
|
||||
'BODY_MODEL_PATH':
|
||||
'ms://damo/scepter_scedit@annotator/ckpts/body_pose_model.pth',
|
||||
'ms://iic/scepter_scedit@annotator/ckpts/body_pose_model.pth',
|
||||
'HAND_MODEL_PATH':
|
||||
'ms://damo/scepter_scedit@annotator/ckpts/hand_pose_model.pth',
|
||||
'ms://iic/scepter_scedit@annotator/ckpts/hand_pose_model.pth',
|
||||
'INPUT_KEYS': ['img'],
|
||||
'OUTPUT_KEYS': ['openpose']
|
||||
}
|
||||
midas_cfg = {
|
||||
'NAME': 'MidasDetector',
|
||||
'PRETRAINED_MODEL':
|
||||
'ms://damo/scepter_scedit@annotator/ckpts/dpt_hybrid-midas-501f0c75.pt',
|
||||
'ms://iic/scepter_scedit@annotator/ckpts/dpt_hybrid-midas-501f0c75.pt',
|
||||
'INPUT_KEYS': ['img'],
|
||||
'OUTPUT_KEYS': ['depth']
|
||||
}
|
||||
mlsd_cfg = {
|
||||
'NAME': 'MLSDdetector',
|
||||
'PRETRAINED_MODEL':
|
||||
'ms://damo/scepter_scedit@annotator/ckpts/mlsd_large_512_fp32.pth',
|
||||
'ms://iic/scepter_scedit@annotator/ckpts/mlsd_large_512_fp32.pth',
|
||||
'INPUT_KEYS': ['img'],
|
||||
'OUTPUT_KEYS': ['mlsd']
|
||||
}
|
||||
|
||||
@@ -3,14 +3,13 @@
|
||||
|
||||
from abc import ABCMeta, abstractmethod
|
||||
|
||||
from torch.utils.data import Dataset
|
||||
|
||||
from scepter.modules.transform.registry import TRANSFORMS, build_pipeline
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
from scepter.modules.utils.distribute import we
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.logger import get_logger
|
||||
from scepter.modules.utils.registry import old_python_version
|
||||
from torch.utils.data import Dataset
|
||||
|
||||
|
||||
class BaseDataset(Dataset, metaclass=ABCMeta):
|
||||
|
||||
@@ -8,7 +8,6 @@ from collections.abc import Iterable
|
||||
|
||||
import numpy as np
|
||||
import torchvision
|
||||
|
||||
from scepter.modules.data.dataset.base_dataset import BaseDataset
|
||||
from scepter.modules.data.dataset.registry import DATASETS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
@@ -101,6 +100,10 @@ class ImageTextPairDataset(BaseDataset):
|
||||
'NEGTIVE_PROMPT': {
|
||||
'value': '',
|
||||
'description': 'The default negtive prompt',
|
||||
},
|
||||
'DATA_NUM': {
|
||||
'value': '',
|
||||
'description': '',
|
||||
}
|
||||
}
|
||||
para_dict.update(BaseDataset.para_dict)
|
||||
@@ -108,6 +111,7 @@ class ImageTextPairDataset(BaseDataset):
|
||||
def __init__(self, cfg, logger=None):
|
||||
super(ImageTextPairDataset, self).__init__(cfg, logger=logger)
|
||||
self.p_zero = cfg.get('P_ZERO', 0.0)
|
||||
self.real_number = cfg.get('DATA_NUM', None)
|
||||
self._default_item = {
|
||||
'meta': {},
|
||||
'prompt':
|
||||
@@ -267,6 +271,11 @@ class Text2ImageDataset(BaseDataset):
|
||||
item['prompt'] = prompt_prefix + value
|
||||
elif key in ['oss_key', 'path', 'img_path', 'target_img_path']:
|
||||
item['meta']['img_path'] = os.path.join(path_prefix, value)
|
||||
elif key in [
|
||||
'src_oss_key', 'src_path', 'src_img_path',
|
||||
'src_target_img_path'
|
||||
]:
|
||||
item['meta']['src_path'] = os.path.join(path_prefix, value)
|
||||
elif key in ['width', 'height']:
|
||||
item['meta'][key] = int(value)
|
||||
else:
|
||||
|
||||