Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
dc8d0abca1 | ||
|
|
034f394a62 | ||
|
|
b1105be87b | ||
|
|
01fd8335af | ||
|
|
7a9f90efb2 | ||
|
|
cbbef4a2da | ||
|
|
b242449b25 | ||
|
|
2fb8fd1872 | ||
|
|
ef82a32944 | ||
|
|
aa7e959330 | ||
|
|
edff6352d5 | ||
|
|
aa8250e5d2 | ||
|
|
e7255aac25 | ||
|
|
b752ff1cbc | ||
|
|
8b08629bd0 | ||
|
|
c70ef0fc47 | ||
|
|
8076aae7da | ||
|
|
79e4910ce4 | ||
|
|
6be496d5ac | ||
|
|
b5b2fd9974 | ||
|
|
f502f8063d | ||
|
|
a3711e1d1d | ||
|
|
8ba0b4c673 | ||
|
|
4f4e516164 | ||
|
|
88f0244516 | ||
|
|
5733918aec | ||
|
|
da88802b68 | ||
|
|
16d99a268f | ||
|
|
fc46a63ac8 | ||
|
|
0010d4282a | ||
|
|
8a9b791e3f | ||
|
|
ee7fe888f2 | ||
|
|
36b259b4b4 | ||
|
|
92c849412e | ||
|
|
2a7e026f84 | ||
|
|
cdba82baf8 | ||
|
|
565c7957d8 | ||
|
|
e00c23d09a | ||
|
|
bf53829530 | ||
|
|
35aada8ce8 | ||
|
|
d3ce651bf7 | ||
|
|
7a58c91940 | ||
|
|
4e1606af2d | ||
|
|
09459c11b7 | ||
|
|
d9a48268d5 | ||
|
|
9f1847501d | ||
|
|
8214227098 | ||
|
|
2e69b2b116 | ||
|
|
3440ec7c38 | ||
|
|
9999e0e1f9 | ||
|
|
01c03683e8 | ||
|
|
9adb273e4b | ||
|
|
2249ff37c9 |
@@ -15,6 +15,6 @@ dev
|
||||
scepter.egg-info
|
||||
.readthedocs.yml
|
||||
1.9
|
||||
MANIFEST.in
|
||||
#MANIFEST.in
|
||||
*resources
|
||||
*.ipynb_checkpoints*
|
||||
|
||||
@@ -1 +1,2 @@
|
||||
recursive-include scepter *.yaml
|
||||
recursive-include scepter *.md
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
|
After Width: | Height: | Size: 4.3 KiB |
|
After Width: | Height: | Size: 8.8 MiB |
|
After Width: | Height: | Size: 2.9 KiB |
|
After Width: | Height: | Size: 90 KiB |
|
After Width: | Height: | Size: 72 KiB |
|
After Width: | Height: | Size: 62 KiB |
|
After Width: | Height: | Size: 66 KiB |
|
After Width: | Height: | Size: 66 KiB |
|
After Width: | Height: | Size: 27 KiB |
|
After Width: | Height: | Size: 87 KiB |
|
After Width: | Height: | Size: 91 KiB |
|
After Width: | Height: | Size: 27 KiB |
|
After Width: | Height: | Size: 72 KiB |
|
After Width: | Height: | Size: 64 KiB |
|
After Width: | Height: | Size: 121 KiB |
|
After Width: | Height: | Size: 16 KiB |
|
After Width: | Height: | Size: 19 KiB |
|
After Width: | Height: | Size: 120 KiB |
|
After Width: | Height: | Size: 117 KiB |
|
After Width: | Height: | Size: 121 KiB |
|
After Width: | Height: | Size: 22 KiB |
|
After Width: | Height: | Size: 39 KiB |
|
After Width: | Height: | Size: 17 KiB |
|
After Width: | Height: | Size: 109 KiB |
|
After Width: | Height: | Size: 247 KiB |
|
After Width: | Height: | Size: 301 KiB |
|
After Width: | Height: | Size: 230 KiB |
|
After Width: | Height: | Size: 353 KiB |
|
After Width: | Height: | Size: 32 KiB |
|
After Width: | Height: | Size: 144 KiB |
|
After Width: | Height: | Size: 126 KiB |
|
After Width: | Height: | Size: 138 KiB |
|
After Width: | Height: | Size: 170 KiB |
|
After Width: | Height: | Size: 273 KiB |
|
After Width: | Height: | Size: 243 KiB |
|
After Width: | Height: | Size: 134 KiB |
|
After Width: | Height: | Size: 119 KiB |
|
After Width: | Height: | Size: 433 KiB |
|
After Width: | Height: | Size: 265 KiB |
|
After Width: | Height: | Size: 300 KiB |
|
After Width: | Height: | Size: 155 KiB |
|
After Width: | Height: | Size: 15 KiB |
|
After Width: | Height: | Size: 20 KiB |
|
After Width: | Height: | Size: 49 KiB |
|
After Width: | Height: | Size: 45 KiB |
|
After Width: | Height: | Size: 34 KiB |
|
After Width: | Height: | Size: 129 KiB |
|
After Width: | Height: | Size: 101 KiB |
|
After Width: | Height: | Size: 100 KiB |
|
After Width: | Height: | Size: 104 KiB |
|
After Width: | Height: | Size: 103 KiB |
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
# Configuration file for the Sphinx documentation builder.
|
||||
#
|
||||
# This file only contains a selection of the most common options. For a full
|
||||
|
||||
@@ -14,8 +14,8 @@ Model modules are divided into backbones, necks, heads, loss, metrics, networks,
|
||||
Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import BACKBONES
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@BACKBONES.register_class("ResNet")
|
||||
@@ -25,8 +25,8 @@ class ResNet(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import NECKS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import NECKS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@NECKS.register_class()
|
||||
@@ -36,8 +36,8 @@ class GlobalAveragePooling(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import HEADS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import HEADS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@HEADS.register_class()
|
||||
@@ -47,7 +47,7 @@ class ClassifierHead(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import LOSSES
|
||||
from scepter.modules.model.registry import LOSSES
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
@@ -59,7 +59,7 @@ class CrossEntropy(nn.Module):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES, NECKS, HEADS, LOSSES
|
||||
from scepter.modules.model.registry import BACKBONES, NECKS, HEADS, LOSSES
|
||||
|
||||
backbone = BACKBONES.build(cfg.BACKBONE, logger=logger)
|
||||
neck = NECKS.build(cfg.NECK, logger=logger)
|
||||
@@ -83,8 +83,8 @@ To be implemented specifically as needed;
|
||||
Basic Usage Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.model.metrics.base_metric import BaseMetric
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.base_metric import BaseMetric
|
||||
|
||||
|
||||
@METRICS.register_class("AccuracyMetric")
|
||||
@@ -95,7 +95,7 @@ class AccuracyMetric(BaseMetric):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
|
||||
metric = METRICS.build(cfgs, logger)
|
||||
```
|
||||
@@ -117,8 +117,8 @@ Typically takes logits and labels as well as other necessary variables as inputs
|
||||
Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.model.tokenizers import BaseTokenizer
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.tokenizers import BaseTokenizer
|
||||
|
||||
|
||||
@TOKENIZERS.register_class()
|
||||
@@ -129,7 +129,7 @@ class BaseBertTokenizer(BaseTokenizer):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
|
||||
tokenizer = TOKENIZERS.build(cfgs, logger)
|
||||
```
|
||||
@@ -147,8 +147,8 @@ Takes a list of texts that need tokenization as input and outputs token id seque
|
||||
Subclass registration:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.model.networks.train_module import TrainModule
|
||||
from scepter.modules.model.registry import MODELS
|
||||
from scepter.modules.model.networks.train_module import TrainModule
|
||||
|
||||
|
||||
@MODELS.register_class()
|
||||
@@ -159,7 +159,7 @@ class Classifier(TrainModule):
|
||||
Actual usage:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.modules.model.registry import MODELS
|
||||
|
||||
model = MODELS.build(self.cfg.MODEL, logger=self.logger)
|
||||
```
|
||||
|
||||
@@ -9,8 +9,8 @@
|
||||
Usage when subclassing lr_schedulers:
|
||||
|
||||
```python
|
||||
from scepter.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
from scepter.modules.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.modules.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
|
||||
|
||||
@LR_SCHEDULERS.register_class()
|
||||
@@ -48,8 +48,8 @@ Sets up the schedule for the passed-in optimizer object;
|
||||
Usage when subclassing optimizers:
|
||||
|
||||
```python
|
||||
from scepter.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.opt.optimizers.registry import OPTIMIZERS
|
||||
from scepter.modules.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.modules.opt.optimizers.registry import OPTIMIZERS
|
||||
|
||||
|
||||
@OPTIMIZERS.register_class()
|
||||
|
||||
@@ -6,17 +6,17 @@ This is the File System Module, designed to handle file transfer functionalities
|
||||
|
||||
The component currently supports three types of IO Handler:
|
||||
|
||||
1. scepter.utils.file_clients.AliyunOssFs
|
||||
2. scepter.utils.file_clients.LocalFs
|
||||
3. scepter.utils.file_clients.HttpFs
|
||||
1. scepter.modules.utils.file_clients.AliyunOssFs
|
||||
2. scepter.modules.utils.file_clients.LocalFs
|
||||
3. scepter.modules.utils.file_clients.HttpFs
|
||||
|
||||
<hr/>
|
||||
|
||||
## Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
fs_cfg = Config(load=False, cfg_dict={
|
||||
"NAME": "AliyunOssFs",
|
||||
|
||||
@@ -4,18 +4,18 @@ Relies on SDKs, which are used to organize modules and SDKs that are frequently
|
||||
|
||||
## Overview
|
||||
|
||||
1. Parameter sdk (scepter.utils.config)
|
||||
2. Path sdk (scepter.utils.directory)
|
||||
3. PyTorch distributed sdk (scepter.utils.distribute)
|
||||
4. Model export sdk (scepter.utils.export_model)
|
||||
5. File system sdk (scepter.utils.file_system)
|
||||
6. Logging sdk (scepter.utils.logger)
|
||||
7. Video processing sdk (scepter.utils.video_reader), see the document (video_reader.md)
|
||||
8. Module registration sdk (scepter.utils.registry)
|
||||
9. Data sdk (scepter.utils.data)
|
||||
10. Model sdk (scepter.utils.model)
|
||||
11. Sampler sdk (scepter.utils.sampler)
|
||||
12. Probing sdk (scepter.utils.probe)
|
||||
1. Parameter sdk (scepter.modules.utils.config)
|
||||
2. Path sdk (scepter.modules.utils.directory)
|
||||
3. PyTorch distributed sdk (scepter.modules.utils.distribute)
|
||||
4. Model export sdk (scepter.modules.utils.export_model)
|
||||
5. File system sdk (scepter.modules.utils.file_system)
|
||||
6. Logging sdk (scepter.modules.utils.logger)
|
||||
7. Video processing sdk (scepter.modules.utils.video_reader), see the document (video_reader.md)
|
||||
8. Module registration sdk (scepter.modules.utils.registry)
|
||||
9. Data sdk (scepter.modules.utils.data)
|
||||
10. Model sdk (scepter.modules.utils.model)
|
||||
11. Sampler sdk (scepter.modules.utils.sampler)
|
||||
12. Probing sdk (scepter.modules.utils.probe)
|
||||
|
||||
<hr/>
|
||||
|
||||
@@ -24,7 +24,7 @@ Relies on SDKs, which are used to organize modules and SDKs that are frequently
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.config import Config
|
||||
# Initialize Config object from a dict
|
||||
fs_cfg = Config(load=False, cfg_dict={"NAME": "LocalFs"})
|
||||
print(fs_cfg.NAME)
|
||||
@@ -105,7 +105,7 @@ print(fs_cfg.args)
|
||||
Some commonly used path functions
|
||||
### Basic Usage
|
||||
```python
|
||||
from scepter.utils.directory import osp_path
|
||||
from scepter.modules.utils.directory import osp_path
|
||||
# Automatically join paths based on the path prefix
|
||||
prefix = "xxxx"
|
||||
data_file = "example_videos/1.mp4"
|
||||
@@ -114,13 +114,13 @@ print(osp_path(prefix, data_file))
|
||||
# Also outputs as xxxx/example_videos/1.mp4
|
||||
data_file = "xxxx/example_videos/1.mp4"
|
||||
print(osp_path(prefix, data_file))
|
||||
from scepter.utils.directory import get_relative_folder
|
||||
from scepter.modules.utils.directory import get_relative_folder
|
||||
# Get the folder path at a specified level according to the path
|
||||
# By default, the last level xxxx/example_videos/
|
||||
print(get_relative_folder(data_file))
|
||||
# The second last level xxxx/
|
||||
print(get_relative_folder(data_file, keep_index=-2))
|
||||
from scepter.utils.directory import get_md5
|
||||
from scepter.modules.utils.directory import get_md5
|
||||
# Get the md5 code of the text/path 34a447fb46d0b786a3999c9dad01d470
|
||||
print(get_md5(data_file))
|
||||
```
|
||||
@@ -175,8 +175,8 @@ PyTorch distributed initialization SDK. By using this SDK, users can avoid focus
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.distribute import we
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.distribute import we
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
cfg = Config(cfg_dict={}, load=False)
|
||||
|
||||
@@ -304,12 +304,12 @@ Since cloning is involved, this may cause additional GPU memory waste.
|
||||
**Returns**
|
||||
- **tensor** —— The output tensor on the CPU for process rank=0.
|
||||
|
||||
## 4. 模型导出sdk(scepter.utils.export_model)
|
||||
## 4. 模型导出sdk(scepter.modules.utils.export_model)
|
||||
APIs for exporting models to TorchScript/ONNX formats.
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.export_model import save_develop_model_multi_io
|
||||
from scepter.modules.utils.export_model import save_develop_model_multi_io
|
||||
|
||||
save_develop_model_multi_io(
|
||||
model,
|
||||
@@ -345,16 +345,16 @@ Supports importing and exporting models with multiple inputs and outputs
|
||||
**Returns**
|
||||
- **tensor** —— The output tensor on the CPU for process rank=0.
|
||||
|
||||
## 5. 文件系统sdk(scepter.utils.file_system)
|
||||
## 5. 文件系统sdk(scepter.modules.utils.file_system)
|
||||
Refer to [file_clients](file_clients.md)
|
||||
|
||||
## 6. Logging SDK(scepter.utils.logger)
|
||||
## 6. Logging SDK(scepter.modules.utils.logger)
|
||||
Used to instantiate a standard logging instance for printing information.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.logger import get_logger, init_logger
|
||||
from scepter.modules.utils.logger import get_logger, init_logger
|
||||
|
||||
std_logger = get_logger(name="scepter")
|
||||
init_logger(std_logger, log_file="", dist_launcher="pytorch")
|
||||
@@ -405,14 +405,14 @@ Calculate the time remaining until completion based on the current usage time an
|
||||
**Returns**
|
||||
- **str** —— Formatted output.
|
||||
|
||||
## 7. Video Processing SDK (scepter.utils.video_reader)
|
||||
## 7. Video Processing SDK (scepter.modules.utils.video_reader)
|
||||
APIs for handling video reading.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.utils.video_reader.video_reader import (
|
||||
from scepter.modules.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.modules.utils.video_reader.video_reader import (
|
||||
VideoReaderWrapper, EasyVideoReader, FramesReaderWrapper
|
||||
)
|
||||
```
|
||||
@@ -554,14 +554,14 @@ Iterator, with each iteration returning a tensor of a segment.
|
||||
**Returns**
|
||||
- **tensor** —— The tensor of the video segment.
|
||||
|
||||
## 8. Module Registration SDK (scepter.utils.registry)
|
||||
## 8. Module Registration SDK (scepter.modules.utils.registry)
|
||||
Used for managing various registered classes.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from scepter.utils.registry import Registry
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.registry import Registry
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
MODELS = Registry('MODELS')
|
||||
|
||||
@@ -614,14 +614,14 @@ Register a function
|
||||
**Returns**
|
||||
- **name** —— Registration name.
|
||||
|
||||
## 9. Data SDK(scepter.utils.data)
|
||||
## 9. Data SDK(scepter.modules.utils.data)
|
||||
Used for transferring data between devices
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
from scepter.modules.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
|
||||
data = {"a": torch.Tensor([0])}
|
||||
transfer_data_to_numpy(data)
|
||||
@@ -668,7 +668,7 @@ Used for operations such as loading and evaluating models
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.model import move_model_to_cpu, load_pretrained,
|
||||
from scepter.modules.utils.model import move_model_to_cpu, load_pretrained,
|
||||
count_params, init_weights
|
||||
```
|
||||
<hr/>
|
||||
@@ -716,14 +716,14 @@ Initialize the parameters of the model modules.
|
||||
**Parameters**
|
||||
- **module** —— The torch.nn.Module model instance.
|
||||
|
||||
## 11. Sampler SDK(scepter.utils.sampler)
|
||||
## 11. Sampler SDK(scepter.modules.utils.sampler)
|
||||
Samplers are quite universal, and in most cases, custom development is not required. Here are provided several common types of sampler.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.sampler import MultiFoldDistributedSampler,
|
||||
from scepter.modules.utils.sampler import MultiFoldDistributedSampler,
|
||||
EvalDistributedSampler, MultiLevelBatchSampler, MixtureOfSamplers
|
||||
```
|
||||
<hr/>
|
||||
@@ -830,17 +830,17 @@ A sampler for multi-level indexing of large-scale data.
|
||||
|
||||
Iterator, each iteration returns an index of a sample.
|
||||
|
||||
## 12. Prober SDK(scepter.utils.probe)
|
||||
## 12. Prober SDK(scepter.modules.utils.probe)
|
||||
Used for probing variable statistics of various components.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import numpy as np
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.utils.config import Config
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.probe import ProbeData
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.probe import ProbeData
|
||||
|
||||
|
||||
class TestModel(BaseModel):
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
<h1 align="center"> Locate, Assign, Refine: Taming Customized Image Inpainting with Text-Subject Guidance </h1>
|
||||
|
||||
<p align="center">
|
||||
<strong>Yulin Pan</strong>
|
||||
·
|
||||
<strong>Chaojie Mao</strong>
|
||||
·
|
||||
<strong>Zeyinzi Jiang</strong>
|
||||
·
|
||||
<strong>Zhen Han</strong>
|
||||
·
|
||||
<strong>Jingfeng Zhang</strong>
|
||||
<br>
|
||||
<a href="https://arxiv.org/abs/2403.19534"><img src="https://img.shields.io/static/v1?label=arXiv&message=LARGen&color=red&logo=arxiv"></a>
|
||||
<a href="https://ali-vilab.github.io/largen-page/"><img src="https://img.shields.io/badge/Page-LARGen-Gree"></a>
|
||||
</p>
|
||||
|
||||
LARGen is a unified image inpainting framework that supports text-guided, subject-guided and text-subject-guided inpainting simutaneously.
|
||||
Four LARGen-based fantastic applications are now supported by SCEPTER Studio:
|
||||
1. Zoom Out
|
||||
2. Virtual Try On
|
||||
3. Text-Guided Inpainting
|
||||
4. Text-Subject-Guided Inpainting
|
||||
|
||||
## Basic Usage
|
||||
|
||||
Here's a demo showcasing the use of LARGen-based functions.
|
||||
<p align="left">
|
||||
<img src="https://raw.githubusercontent.com/ali-vilab/largen-page/main/public/images/largen.gif" width="1300">
|
||||
</p>
|
||||
|
||||
## Gallery
|
||||
|
||||
### LAR-Gen: Zoom Out
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a temple on fire</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
<td><strong>Zoom-Out</strong><br>CenterAround:0.75</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_scene_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out1.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out2.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out3.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/zoom_out/ex1_zoom_out4.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Virtual Try-on
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Model Image</strong></td>
|
||||
<td><strong>Model Mask</strong></td>
|
||||
<td><strong>Clothing Image</strong></td>
|
||||
<td><strong>Clothing Mask</strong></td>
|
||||
<td><strong>Try-on Output</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/virtual_try_on/model.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/ex2_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/tshirt.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/ex2_subject_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/virtual_try_on/try_on_out.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Inpainting (Text guided)
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a blue and white porcelain</td>
|
||||
<td><strong>Inpainting Mask1</strong></td>
|
||||
<td><strong>Inpainting Output1</strong></td>
|
||||
<td><strong>Inpainting Mask2</strong><br>Prompt: a clock</td>
|
||||
<td><strong>Inpainting Output2</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/inpainting_text/ex3_scene_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/ex3_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/inpainting_text.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/ex3_scene_mask2.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text/inpainting_text2.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### LAR-Gen: Inpainting (Text and Subject guided)
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Prompt: a dog wearing sunglasses</td>
|
||||
<td><strong>Origin Mask</strong></td>
|
||||
<td><strong>Reference Image</strong></td>
|
||||
<td><strong>Reference Mask</strong></td>
|
||||
<td><strong>Inpainting Output</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_scene_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_scene_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_subject_im.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/ex4_subject_mask.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/inpainting_text_ref/inpainting_text_ref.jpg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
## Features
|
||||
|
||||
| **Model** | **Locate** | **Assign** | **Refine** |
|
||||
|:---------:|:----------:|:----------:|:----------:|
|
||||
| SD v1.5 | ⏳ | ⏳ | ⏳ |
|
||||
| SD XL | 🪄 | 🪄 | ⏳ |
|
||||
|
||||
- 🪄 denotes that the feature has been supported.
|
||||
- ⏳ denotes that the feature has not been integrated currently.
|
||||
|
||||
|
||||
## Pretrained Models
|
||||
|
||||
| **Model** | **URL** |
|
||||
|:----------:|:-------:|
|
||||
| largen-sdxl-s22k | [ModelScope](https://www.modelscope.cn/models/iic/LARGEN/summary) |
|
||||
|
||||
|
||||
## BibTeX
|
||||
If our work is useful for your research, please consider citing:
|
||||
```bibtex
|
||||
@article{pan2024locate,
|
||||
title={Locate, Assign, Refine: Taming Customized Image Inpainting with Text-Subject Guidance},
|
||||
author={Pan, Yulin and Mao, Chaojie and Jiang, Zeyinzi and Han, Zhen and Zhang, Jingfeng},
|
||||
journal={arXiv preprint arXiv:2403.19534},
|
||||
year={2024}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,125 @@
|
||||
<p align="center">
|
||||
|
||||
<h2 align="center">SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing</h2>
|
||||
<h3 align="center">(CVPR 2024 Highlight)</h3>
|
||||
<p align="center">
|
||||
<strong>Zeyinzi Jiang</strong>
|
||||
·
|
||||
<strong>Chaojie Mao</strong>
|
||||
·
|
||||
<strong>Yulin Pan</strong>
|
||||
·
|
||||
<strong>Zhen Han</strong>
|
||||
·
|
||||
<strong>Jingfeng Zhang</strong>
|
||||
<br>
|
||||
<b>Alibaba Group</b>
|
||||
<br>
|
||||
<a href="https://arxiv.org/abs/2312.11392"><img src='https://img.shields.io/badge/arXiv-SCEdit-red' alt='Paper PDF'></a>
|
||||
<a href='https://scedit.github.io/'><img src='https://img.shields.io/badge/Project_Page-SCEdit-green' alt='Project Page'></a>
|
||||
<a href='https://github.com/modelscope/scepter'><img src='https://img.shields.io/badge/scepter-SCEdit-yellow'></a>
|
||||
<a href='https://github.com/modelscope/swift'><img src='https://img.shields.io/badge/swift-SCEdit-blue'></a>
|
||||
<br>
|
||||
</p>
|
||||
|
||||
SCEdit is an efficient generative fine-tuning framework proposed by Alibaba TongYi Vision Intelligence Lab. This framework enhances the fine-tuning capabilities for text-to-image generation downstream tasks and enables quick adaptation to specific generative scenarios, **saving 30%-50% of training memory costs compared to LoRA**. Furthermore, it can be directly extended to controllable image generation tasks, **requiring only 7.9% of the parameters that ControlNet needs for conditional generation and saving 30% of memory usage**. It supports various conditional generation tasks including edge maps, depth maps, segmentation maps, poses, color maps, and image completion.
|
||||
|
||||
## Usage
|
||||
|
||||
### Text-to-Image Generation
|
||||
```shell
|
||||
# SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i.yaml
|
||||
# SD v2.1
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd21_768_sce_t2i.yaml
|
||||
# SD XL
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i.yaml
|
||||
```
|
||||
|
||||
### Controllable Image Synthesis
|
||||
```shell
|
||||
# SD v1.5 + hed
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd15_512_sce_ctr_hed.yaml
|
||||
# SD v2.1 + canny
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml
|
||||
# SD XL + depth
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_depth.yaml
|
||||
```
|
||||
|
||||
### Gradio
|
||||
```shell
|
||||
python -m scepter.tools.webui # Then click [Use Tuners] or [Use Controller]
|
||||
```
|
||||
|
||||
## Models
|
||||
|
||||
### Model URL
|
||||
|
||||
| Model | URL |
|
||||
|--------|-------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| SCEdit | [ModelScope](https://modelscope.cn/models/iic/scepter_scedit/summary) [HuggingFace](https://huggingface.co/scepter-studio/scepter_scedit) |
|
||||
|
||||
### Text-to-Image Generation
|
||||
|
||||
| **Model** | **SCEdit** |
|
||||
|:---------:|:----------:|
|
||||
| SD 1.5 | 🪄 |
|
||||
| SD 2.1 | 🪄 |
|
||||
| SD XL | 🪄 |
|
||||
|
||||
### Controllable Image Synthesis
|
||||
|
||||
| **Model** | **Canny** | **HED** | **Depth** | **Pose** | **Color** |
|
||||
|:---------:|:---------:|:-------:|:---------:|:--------:|:---------:|
|
||||
| SD 2.1 | 🪄 | 🪄 | 🪄 | 🪄 | 🪄 |
|
||||
| SD XL | 🪄 | 🪄 | 🪄 | 🪄 | 🪄 |
|
||||
|
||||
|
||||
## Application Gallery
|
||||
|
||||
### Dragon Year Special: Dragon Tuner
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Gold Dragon Tuner</strong></td>
|
||||
<td><strong>Sloppy Dragon Tuner</strong></td>
|
||||
<td><strong>Red Dragon Tuner</strong><br> + Papercraft Mantra</td>
|
||||
<td><strong>Azure Dragon Tuner</strong><br> + Pose Control</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/scedit/tuner_gold_dragon.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/tuner_sloppy_dragon.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/tuner_mantra_papercraft_dragon.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/tuner_pose.jpeg" width="300"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
### Text Effect Image
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Conditional Image</strong></td>
|
||||
<td><strong>Midas Control</strong><br>"Race track, top view"</td>
|
||||
<td><strong>Midas Control</strong><br> + Watercolor Mantra<br>"white lilies"</td>
|
||||
<td><strong>Midas Control</strong><br> + Dragon Tuner<br>"Spring Festival, Chinese dragon"</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/scedit/word_condition.png" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/word_race.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/word_lilies.jpeg" width="300"></td>
|
||||
<td><img src="../../../asset/images/scedit/word_festival.jpeg" width="300"></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
|
||||
|
||||
## BibTeX
|
||||
|
||||
```bibtex
|
||||
@article{jiang2023scedit,
|
||||
title = {SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing},
|
||||
author = {Jiang, Zeyinzi and Mao, Chaojie and Pan, Yulin and Han, Zhen and Zhang, Jingfeng},
|
||||
year = {2023},
|
||||
journal = {arXiv preprint arXiv:2312.11392}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,130 @@
|
||||
|
||||
# StyleBooth: Image Style Editing with Multimodal Instruction
|
||||
|
||||
Zhen Han, Chaojie Mao, Zeyinzi Jiang, Yulin Pan, Jingfeng Zhang
|
||||
|
||||
Alibaba Group
|
||||
|
||||
[[paper](https://arxiv.org/abs/2404.12154)][[Model](https://modelscope.cn/models/iic/stylebooth/summary)] [[Dataset](https://modelscope.cn/models/iic/stylebooth/summary)]
|
||||
|
||||
## Abstract
|
||||
|
||||
Given an original image, image editing aims to generate an image that align with the provided instruction. The challenges are to accept multimodal inputs as instructions and a scarcity of high-quality training data, including crucial triplets of source/target image pairs and multimodal (text and image) instructions. In this paper, we focus on image style editing and present <strong>StyleBooth</strong>, a method that proposes a comprehensive framework for image editing and a feasible strategy for building a high-quality style editing dataset. We integrate encoded textual instruction and image exemplar as a unified condition for diffusion model, enabling the editing of original image following <strong>multimodal instructions</strong>. Furthermore, by <strong>iterative style-destyle tuning and editing</strong> and usability filtering, the StyleBooth dataset provides content-consistent stylized/plain image pairs in various categories of styles. To show the flexibility of StyleBooth, we conduct experiments on diverse tasks, such as textbased style editing, exemplar-based style editing and compositional style editing. The results demonstrate that the quality and variety of training data significantly enhance the ability to preserve content and improve the overall quality of generated images in editing tasks.
|
||||

|
||||
|
||||
## Gallery
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong><br>Gold Dragon Tuner</td>
|
||||
<td><strong>Graffiti Art</strong></td>
|
||||
<td><strong>Adorable Kawaii</strong></td>
|
||||
<td><strong>game-retro game</strong></td>
|
||||
<td><strong>Vincent van Gogh</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/scedit/tuner_gold_dragon.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/graffiti.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/kawaii.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/retrogame.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/vangogh.jpeg" width="240"></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>Origin Image</strong></td>
|
||||
<td><strong>Lowpoly</strong></td>
|
||||
<td><strong>Colored Pencil Art</strong></td>
|
||||
<td><strong>Watercolor</strong></td>
|
||||
<td><strong>misc-disco</strong></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><img src="../../../asset/images/stylebooth/mountain.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/lowpoly.jpg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/colorpencil.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/watercolor.jpeg" width="240"></td>
|
||||
<td><img src="../../../asset/images/stylebooth/disco.jpeg" width="240"></td>
|
||||
</tr>
|
||||
</table>
|
||||
## Features
|
||||
|
||||
| **Text-Based** | **Exemplar-Based** |
|
||||
|:--------------:|:-----------------:|
|
||||
| 🪄 | ⏳ |
|
||||
|
||||
- ✅ indicates support for both training and inference.
|
||||
- 🪄 denotes that the model has been published.
|
||||
- ⏳ denotes that the module has not been integrated currently.
|
||||
- More models will be released in the future.
|
||||
|
||||
## Run StyleBooth
|
||||
- Code implementation: See model configuration and code based on [🪄SCEPTER](https://github.com/modelscope/scepter/blob/main/docs/en/tasks/stylebooth.md).
|
||||
|
||||
- Demo: Try [🖥️SCEPTER Studio](https://github.com/modelscope/scepter/tree/main?tab=readme-ov-file#%EF%B8%8F-scepter-studio).
|
||||
|
||||
- Easy run:
|
||||
Try the following example script to run StyleBooth modified from [tests/modules/test_diffusion_inference.py](https://github.com/modelscope/scepter/blob/main/tests/modules/test_diffusion_inference.py):
|
||||
|
||||
```python
|
||||
# `pip install scepter>0.0.4` or
|
||||
# clone newest SCEPTER and run `PYTHONPATH=./ python <this_script>` at the main branch root.
|
||||
import os
|
||||
import unittest
|
||||
|
||||
from PIL import Image
|
||||
from torchvision.utils import save_image
|
||||
|
||||
from scepter.modules.inference.stylebooth_inference import StyleboothInference
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.logger import get_logger
|
||||
|
||||
|
||||
class DiffusionInferenceTest(unittest.TestCase):
|
||||
def setUp(self):
|
||||
print(('Testing %s.%s' % (type(self).__name__, self._testMethodName)))
|
||||
self.logger = get_logger(name='scepter')
|
||||
config_file = 'scepter/methods/studio/scepter_ui.yaml'
|
||||
cfg = Config(cfg_file=config_file)
|
||||
if 'FILE_SYSTEM' in cfg:
|
||||
for fs_info in cfg['FILE_SYSTEM']:
|
||||
FS.init_fs_client(fs_info)
|
||||
self.tmp_dir = './cache/save_data/diffusion_inference'
|
||||
if not os.path.exists(self.tmp_dir):
|
||||
os.makedirs(self.tmp_dir)
|
||||
|
||||
def tearDown(self):
|
||||
super().tearDown()
|
||||
|
||||
# uncomment this line to skip this module.
|
||||
# @unittest.skip('')
|
||||
def test_stylebooth(self):
|
||||
config_file = 'scepter/methods/studio/inference/edit/stylebooth_tb_pro.yaml'
|
||||
cfg = Config(cfg_file=config_file)
|
||||
diff_infer = StyleboothInference(logger=self.logger)
|
||||
diff_infer.init_from_cfg(cfg)
|
||||
|
||||
output = diff_infer({'prompt': 'Let this image be in the style of sai-lowpoly'},
|
||||
style_edit_image=Image.open('asset/images/inpainting_text_ref/ex4_scene_im.jpg'),
|
||||
style_guide_scale_text=7.5,
|
||||
style_guide_scale_image=1.5)
|
||||
save_path = os.path.join(self.tmp_dir,
|
||||
'stylebooth_test_lowpoly_cute_dog.png')
|
||||
save_image(output['images'], save_path)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
```
|
||||
|
||||
## StyleTuner and De-StyleTuner.
|
||||
|
||||
### Base I2I Model.
|
||||
|
||||
For style and de-style tuning, we use a private high-resolution I2I model trained with [InstructPix2Pix dataset](https://instruct-pix2pix.eecs.berkeley.edu/) as base model. However, one can try the same tunning process using this [yaml](https://github.com/modelscope/scepter/blob/main/scepter/methods/edit/edit_512_lora.yaml) based on any other I2I model, such as StyleBooth (shown in this yaml), [InstructPix2Pix](https://github.com/timothybrooks/instruct-pix2pix) or [MagicBrush](https://github.com/OSU-NLP-Group/MagicBrush).
|
||||
|
||||
### Training Data.
|
||||
|
||||
Please check the zips for correct format: [De-Text](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fdetext.zip), [Image2Hed](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fhed_pair.zip), [Image2Depth](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fimage2depth.zip), [Depth2Image](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fdepth2image.zip).
|
||||
|
||||
### Launch.
|
||||
|
||||
See [code](https://github.com/modelscope/scepter/blob/cbbef4a2da5b66fc33b9f8ece7f2fb4aac9d6e3c/tests/tools/test_train.py#L220) for more information.
|
||||
@@ -0,0 +1,31 @@
|
||||
<h1 align="center">Dataset Management</h1>
|
||||
|
||||
SCEPTER supports three types of dataset formats: TXT, CSV, and ModelScope.
|
||||
Below are examples for each format, illustrating their details and basic usage.
|
||||
|
||||
## Modelscope Format
|
||||
|
||||
We use a [custom-stylized dataset](https://modelscope.cn/datasets/iic/style_custom_dataset/summary), which included classes 3D, anime, flat illustration, oil painting, sketch, and watercolor, each with 30 image-text pairs.
|
||||
|
||||
```python
|
||||
# pip install modelscope
|
||||
from modelscope.msdatasets import MsDataset
|
||||
ms_train_dataset = MsDataset.load('style_custom_dataset', namespace='damo', subset_name='3D', split='train_short')
|
||||
print(next(iter(ms_train_dataset)))
|
||||
```
|
||||
|
||||
## CSV Format
|
||||
|
||||
For the data format used by SCEPTER Studio, please refer to [3D_example_csv.zip](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_csv.zip) and [hed_pair.zip](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets%2Fhed_pair.zip).
|
||||
```shell
|
||||
mkdir -p cache/datasets/ && wget 'https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_csv.zip' -O cache/datasets/3D_example_csv.zip && unzip cache/datasets/3D_example_csv.zip -d cache/datasets/ && rm cache/datasets/3D_example_csv.zip
|
||||
mkdir -p cache/datasets/ && wget 'https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/hed_pair.zip' -O cache/datasets/hed_pair.zip && unzip cache/datasets/hed_pair.zip -d cache/datasets/ && rm cache/datasets/hed_pair.zip
|
||||
```
|
||||
|
||||
## TXT Format
|
||||
|
||||
To facilitate starting training in command-line mode, you can use a dataset in text format, please refer to [3D_example_txt.zip](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_txt.zip)
|
||||
|
||||
```shell
|
||||
mkdir -p cache/datasets/ && wget 'https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=datasets/3D_example_txt.zip' -O cache/datasets/3D_example_txt.zip && unzip cache/datasets/3D_example_txt.zip -d cache/datasets/ && rm cache/datasets/3D_example_txt.zip
|
||||
```
|
||||
@@ -0,0 +1,45 @@
|
||||
# Inference
|
||||
|
||||
In this tutorial, we'll cover the use of the scepter framework for convenient inference, including inference using the command line or specific method classes, and we'll give examples of inference methods for additional tasks.
|
||||
|
||||
## Command Line
|
||||
Inference of SDXL generation models using the command line.
|
||||
```shell
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_xl_1024.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD XL
|
||||
```
|
||||
|
||||
## Class Instantiation
|
||||
Inference of SD2.1 generation models using the class instantiation.
|
||||
```python
|
||||
from torchvision.utils import save_image
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.logger import get_logger
|
||||
from scepter.modules.inference.diffusion_inference import DiffusionInference
|
||||
# init file system - modelscope
|
||||
FS.init_fs_client(Config(load=False, cfg_dict={'NAME': 'ModelscopeFs', 'TEMP_DIR': 'cache/data'}))
|
||||
# init model config
|
||||
logger = get_logger(name='scepter')
|
||||
cfg = Config(cfg_file='scepter/methods/studio/inference/stable_diffusion/sd21_pro.yaml')
|
||||
diff_infer = DiffusionInference(logger)
|
||||
diff_infer.init_from_cfg(cfg)
|
||||
# start inference
|
||||
output = diff_infer({'prompt': 'a cute dog'})
|
||||
save_image(output['images'], 'sd21_test_prompt_a_cute_dog.png')
|
||||
```
|
||||
|
||||
## Additional Tasks
|
||||
|
||||
### Fine-tuned Model Inference
|
||||
|
||||
```shell
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i_swift.yaml --pretrained_model 'cache/save_data/sd15_512_sce_t2i_swift/checkpoints/ldm_step-100.pth' --prompt 'A close up of a small rabbit wearing a hat and scarf' --save_folder 'trained_test_prompt_rabbit'
|
||||
```
|
||||
|
||||
### Controllable Image Synthesis Inference
|
||||
|
||||
- SCEdit
|
||||
```shell
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml --num_samples 1 --prompt 'a single flower is shown in front of a tree' --save_folder 'test_flower_canny' --image_size 768 --task control --image 'asset/images/flower.jpg' --control_mode canny --pretrained_model ms://iic/scepter_scedit@controllable_model/SD2.1/canny_control/0_SwiftSCETuning/pytorch_model.bin # canny
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml --num_samples 1 --prompt 'super mario' --save_folder 'test_mario_pose' --image_size 768 --task control --image 'asset/images/pose_source.png' --control_mode source --pretrained_model ms://iic/scepter_scedit@controllable_model/SD2.1/pose_control/0_SwiftSCETuning/pytorch_model.bin # pose
|
||||
```
|
||||
@@ -0,0 +1,84 @@
|
||||
# Training
|
||||
|
||||
We provide a framework for training and validation.
|
||||
|
||||
The scripts below are just for illustration purposes. To achieve better results, you can modify the corresponding parameters as needed.
|
||||
|
||||
## Start Training
|
||||
There are different ways to start a training:
|
||||
|
||||
- calling scepter/tools/run_train.py:
|
||||
```bash
|
||||
# calling at SCEPTER root:
|
||||
PYTHONPATH=./ python scepter/tools/run_train.py --cfg [path-to-your-yaml]
|
||||
|
||||
# calling scepter library:
|
||||
pip install scepter
|
||||
python -m scepter.tools.run_train --cfg [path-to-your-yaml]
|
||||
```
|
||||
- calling your own script:
|
||||
```bash
|
||||
# calling at SCEPTER root:
|
||||
PYTHONPATH=./ python [path-to-your-script] --cfg [path-to-your-yaml]
|
||||
|
||||
# calling scepter library:
|
||||
pip install scepter
|
||||
python [path-to-your-script] --cfg [path-to-your-yaml]
|
||||
```
|
||||
your scepter should be like:
|
||||
```python
|
||||
from scepter.tools.run_train import run
|
||||
|
||||
if __name__ == '__main__':
|
||||
run()
|
||||
```
|
||||
|
||||
## Popular Tasks
|
||||
### Text-to-Image Generation
|
||||
|
||||
- SCEdit
|
||||
```bash
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i.yaml # SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd21_768_sce_t2i.yaml # SD v2.1
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i.yaml # SD XL
|
||||
```
|
||||
|
||||
- Existing Tuning Strategies
|
||||
```bash
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_1.5_512.yaml # fully-tuning on SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_2.1_768_lora.yaml # lora-tuning on SD v2.1
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```bash
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i_datatxt.yaml
|
||||
```
|
||||
|
||||
### Controllable Image Synthesis
|
||||
|
||||
- SCEdit
|
||||
|
||||
The YAML configuration can be modified to combine different base models and conditions. The following is provided as an example.
|
||||
```bash
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd15_512_sce_ctr_hed.yaml # SD v1.5 + hed
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml # SD v2.1 + canny
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml # SD v2.1 + pose
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_depth.yaml # SD XL + depth
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color.yaml # SD XL + color
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```bash
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color_datatxt.yaml
|
||||
```
|
||||
|
||||
|
||||
## Customize Modules
|
||||
You can register your own Modules like DATASET, SAMPLERS, TRANSFORMS, MODELS, SOVLERS, HOOKS, OPTIMIZERS into SCEPTER.
|
||||
Refer to `example/`, build the modules of your task in `example/{task}`.
|
||||
```bash
|
||||
cd example/classifier
|
||||
python run.py --cfg classifier.yaml
|
||||
```
|
||||
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
# Configuration file for the Sphinx documentation builder.
|
||||
#
|
||||
# This file only contains a selection of the most common options. For a full
|
||||
|
||||
@@ -15,8 +15,8 @@
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import BACKBONES
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@BACKBONES.register_class("ResNet")
|
||||
@@ -26,8 +26,8 @@ class ResNet(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import NECKS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import NECKS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@NECKS.register_class()
|
||||
@@ -37,8 +37,8 @@ class GlobalAveragePooling(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import HEADS
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.modules.model.registry import HEADS
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
|
||||
|
||||
@HEADS.register_class()
|
||||
@@ -48,7 +48,7 @@ class ClassifierHead(BaseModel):
|
||||
```
|
||||
|
||||
```python
|
||||
from scepter.model.registry import LOSSES
|
||||
from scepter.modules.model.registry import LOSSES
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
@@ -60,7 +60,7 @@ class CrossEntropy(nn.Module):
|
||||
实际调用:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import BACKBONES, NECKS, HEADS, LOSSES, TUNERS
|
||||
from scepter.modules.model.registry import BACKBONES, NECKS, HEADS, LOSSES, TUNERS
|
||||
|
||||
backbone = BACKBONES.build(cfg.BACKBONE, logger=logger)
|
||||
neck = NECKS.build(cfg.NECK, logger=logger)
|
||||
@@ -85,8 +85,8 @@ tuner = TUNERS.build(cfg.TUNER, logger=logger)
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.model.metrics.base_metric import BaseMetric
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.base_metric import BaseMetric
|
||||
|
||||
|
||||
@METRICS.register_class("AccuracyMetric")
|
||||
@@ -97,7 +97,7 @@ class AccuracyMetric(BaseMetric):
|
||||
实际用法:
|
||||
|
||||
```python
|
||||
from scepter.model.metrics.registry import METRICS
|
||||
from scepter.modules.model.metrics.registry import METRICS
|
||||
|
||||
metric = METRICS.build(cfgs, logger)
|
||||
```
|
||||
@@ -119,8 +119,8 @@ metric = METRICS.build(cfgs, logger)
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.model.tokenizers import BaseTokenizer
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.tokenizers import BaseTokenizer
|
||||
|
||||
|
||||
@TOKENIZERS.register_class()
|
||||
@@ -131,7 +131,7 @@ class BaseBertTokenizer(BaseTokenizer):
|
||||
实际用法:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import TOKENIZERS
|
||||
from scepter.modules.model.registry import TOKENIZERS
|
||||
|
||||
tokenizer = TOKENIZERS.build(cfgs, logger)
|
||||
```
|
||||
@@ -149,8 +149,8 @@ tokenizer = TOKENIZERS.build(cfgs, logger)
|
||||
子类注册:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.model.networks.train_module import TrainModule
|
||||
from scepter.modules.model.registry import MODELS
|
||||
from scepter.modules.model.networks.train_module import TrainModule
|
||||
|
||||
|
||||
@MODELS.register_class()
|
||||
@@ -161,7 +161,7 @@ class Classifier(TrainModule):
|
||||
实际用法:
|
||||
|
||||
```python
|
||||
from scepter.model.registry import MODELS
|
||||
from scepter.modules.model.registry import MODELS
|
||||
|
||||
model = MODELS.build(self.cfg.MODEL, logger=self.logger)
|
||||
```
|
||||
|
||||
@@ -9,8 +9,8 @@
|
||||
子lr_schedulers继承时用法:
|
||||
|
||||
```python
|
||||
from scepter.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
from scepter.modules.opt.lr_schedulers import LR_SCHEDULERS
|
||||
from scepter.modules.opt.lr_schedulers.base_scheduler import BaseScheduler
|
||||
|
||||
|
||||
@LR_SCHEDULERS.register_class()
|
||||
@@ -48,8 +48,8 @@ lr_schedulers的基类,支持注册操作,可根据需要自定义;
|
||||
子optimizers继承时用法:
|
||||
|
||||
```python
|
||||
from scepter.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.opt.optimizers.registry import OPTIMIZERS
|
||||
from scepter.modules.opt.optimizers.base_optimizer import BaseOptimize
|
||||
from scepter.modules.opt.optimizers.registry import OPTIMIZERS
|
||||
|
||||
|
||||
@OPTIMIZERS.register_class()
|
||||
|
||||
@@ -6,10 +6,10 @@
|
||||
|
||||
支持3类文件IO Handler:
|
||||
|
||||
1. scepter.utils.file_clients.AliyunOssFs
|
||||
2. scepter.utils.file_clients.LocalFs
|
||||
3. scepter.utils.file_clients.HttpFs
|
||||
4. scepter.utils.file_clients.ModelscopeFs
|
||||
1. scepter.modules.utils.file_clients.AliyunOssFs
|
||||
2. scepter.modules.utils.file_clients.LocalFs
|
||||
3. scepter.modules.utils.file_clients.HttpFs
|
||||
4. scepter.modules.utils.file_clients.ModelscopeFs
|
||||
|
||||
|
||||
<hr/>
|
||||
@@ -17,8 +17,8 @@
|
||||
## 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
fs_cfg = Config(load=False, cfg_dict={
|
||||
"NAME": "AliyunOssFs",
|
||||
|
||||
@@ -3,18 +3,18 @@
|
||||
依赖SDK,该部分用于对框架全局经常复用的模块和sdk进行整理,并根据功能相关性进行聚合。
|
||||
|
||||
## 总览
|
||||
1. 参数sdk(scepter.utils.config)
|
||||
2. 路径sdk(scepter.utils.directory)
|
||||
3. torch分布式sdk(scepter.utils.distribute)
|
||||
4. 模型导出sdk(scepter.utils.export_model)
|
||||
5. 文件系统sdk(scepter.utils.file_system)
|
||||
6. 日志sdk(scepter.utils.logger)
|
||||
7. 视频处理sdk(scepter.utils.video_reader),文档参考(video_reader.md)
|
||||
8. 模块注册sdk(scepter.utils.registry)
|
||||
9. 数据sdk(scepter.utils.data)
|
||||
10. 模型sdk(scepter.utils.model)
|
||||
11. 采样器sdk(scepter.utils.sampler)
|
||||
12. 探针器sdk(scepter.utils.probe)
|
||||
1. 参数sdk(scepter.modules.utils.config)
|
||||
2. 路径sdk(scepter.modules.utils.directory)
|
||||
3. torch分布式sdk(scepter.modules.utils.distribute)
|
||||
4. 模型导出sdk(scepter.modules.utils.export_model)
|
||||
5. 文件系统sdk(scepter.modules.utils.file_system)
|
||||
6. 日志sdk(scepter.modules.utils.logger)
|
||||
7. 视频处理sdk(scepter.modules.utils.video_reader),文档参考(video_reader.md)
|
||||
8. 模块注册sdk(scepter.modules.utils.registry)
|
||||
9. 数据sdk(scepter.modules.utils.data)
|
||||
10. 模型sdk(scepter.modules.utils.model)
|
||||
11. 采样器sdk(scepter.modules.utils.sampler)
|
||||
12. 探针器sdk(scepter.modules.utils.probe)
|
||||
|
||||
<hr/>
|
||||
|
||||
@@ -23,7 +23,7 @@
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
# 从一个dict对象 初始化 Config对象
|
||||
fs_cfg = Config(load=False, cfg_dict={"NAME": "LocalFs"})
|
||||
@@ -97,7 +97,7 @@ print(fs_cfg.args)
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.directory import osp_path
|
||||
from scepter.modules.utils.directory import osp_path
|
||||
|
||||
# 根据路径前缀进行自动化路径拼接
|
||||
prefix = "xxxx"
|
||||
@@ -108,7 +108,7 @@ print(osp_path(prefix, data_file))
|
||||
data_file = "xxxx/example_videos/1.mp4"
|
||||
print(osp_path(prefix, data_file))
|
||||
|
||||
from scepter.utils.directory import get_relative_folder
|
||||
from scepter.modules.utils.directory import get_relative_folder
|
||||
|
||||
# 根据路径获取指定层级的文件夹路径
|
||||
# 默认最后一级 xxxx/example_videos/
|
||||
@@ -116,7 +116,7 @@ print(get_relative_folder(data_file))
|
||||
# 倒数第二级 xxxx/
|
||||
print(get_relative_folder(data_file, keep_index=-2))
|
||||
|
||||
from scepter.utils.directory import get_md5
|
||||
from scepter.modules.utils.directory import get_md5
|
||||
|
||||
# 获取文本/路径的md5码 34a447fb46d0b786a3999c9dad01d470
|
||||
print(get_md5(data_file))
|
||||
@@ -172,8 +172,8 @@ torch分布式初始化sdk,使用该sdk,可以让用户不要关注torch的
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.distribute import we
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.distribute import we
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
cfg = Config(cfg_dict={}, load=False)
|
||||
|
||||
@@ -304,12 +304,12 @@ we.init_env(cfg, fn, logger=None)
|
||||
**Returns**
|
||||
- **tensor** —— 输出的在进程rank=0上的cpu的tensor。
|
||||
|
||||
## 4. 模型导出sdk(scepter.utils.export_model)
|
||||
## 4. 模型导出sdk(scepter.modules.utils.export_model)
|
||||
用于模型导出为torchscript/Onnx格式的api。
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.export_model import save_develop_model_multi_io
|
||||
from scepter.modules.utils.export_model import save_develop_model_multi_io
|
||||
|
||||
save_develop_model_multi_io(
|
||||
model,
|
||||
@@ -347,16 +347,16 @@ input_type 一一对应。
|
||||
**Returns**
|
||||
- **tensor** —— 输出的在进程rank=0上的cpu的tensor。
|
||||
|
||||
## 5. 文件系统sdk(scepter.utils.file_system)
|
||||
## 5. 文件系统sdk(scepter.modules.utils.file_system)
|
||||
参考[file_clients](file_clients.md)
|
||||
|
||||
## 6. 日志sdk(scepter.utils.logger)
|
||||
## 6. 日志sdk(scepter.modules.utils.logger)
|
||||
用于实例化一个标准的日志实例,用于打印信息。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.logger import get_logger, init_logger
|
||||
from scepter.modules.utils.logger import get_logger, init_logger
|
||||
|
||||
std_logger = get_logger(name="scepter")
|
||||
init_logger(std_logger, log_file="", dist_launcher="pytorch")
|
||||
@@ -407,14 +407,14 @@ init_logger(std_logger, log_file="", dist_launcher="pytorch")
|
||||
**Returns**
|
||||
- **str** —— 格式化的输出。
|
||||
|
||||
## 7. 视频处理sdk(scepter.utils.video_reader)
|
||||
## 7. 视频处理sdk(scepter.modules.utils.video_reader)
|
||||
用于处理视频读取的api。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.utils.video_reader.video_reader import (
|
||||
from scepter.modules.utils.video_reader.frame_sampler import do_frame_sample
|
||||
from scepter.modules.utils.video_reader.video_reader import (
|
||||
VideoReaderWrapper, EasyVideoReader, FramesReaderWrapper
|
||||
)
|
||||
```
|
||||
@@ -556,14 +556,14 @@ overlap: Union[float, Fraction, str] = Fraction(0), transforms: Optional[Callabl
|
||||
**Returns**
|
||||
- **tensor** —— 视频片段的tensor。
|
||||
|
||||
## 8. 模块注册sdk(scepter.utils.registry)
|
||||
## 8. 模块注册sdk(scepter.modules.utils.registry)
|
||||
用于管理各种注册的类。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
from scepter.utils.registry import Registry
|
||||
from scepter.utils.config import Config
|
||||
from scepter.modules.utils.registry import Registry
|
||||
from scepter.modules.utils.config import Config
|
||||
|
||||
MODELS = Registry('MODELS')
|
||||
|
||||
@@ -616,14 +616,14 @@ build目标类的实例
|
||||
**Returns**
|
||||
- **name** —— 注册名称。
|
||||
|
||||
## 9. 数据sdk(scepter.utils.data)
|
||||
## 9. 数据sdk(scepter.modules.utils.data)
|
||||
用于数据在设备间转移
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
from scepter.modules.utils.data import transfer_data_to_numpy, transfer_data_to_cpu, transfer_data_to_cuda
|
||||
|
||||
data = {"a": torch.Tensor([0])}
|
||||
transfer_data_to_numpy(data)
|
||||
@@ -670,7 +670,7 @@ transfer_data_to_cuda(data)
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.model import move_model_to_cpu, load_pretrained,
|
||||
from scepter.modules.utils.model import move_model_to_cpu, load_pretrained,
|
||||
count_params, init_weights
|
||||
```
|
||||
<hr/>
|
||||
@@ -718,14 +718,14 @@ from scepter.utils.model import move_model_to_cpu, load_pretrained,
|
||||
**Parameters**
|
||||
- **module** —— torch.nn.Module模型实例。
|
||||
|
||||
## 11. 采样器sdk(scepter.utils.sampler)
|
||||
## 11. 采样器sdk(scepter.modules.utils.sampler)
|
||||
采样器比较具有通用性,大多数情况下不会进行定制开发,这里提供了几类常用的sampler采样器。
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
import torch
|
||||
from scepter.utils.sampler import MultiFoldDistributedSampler,
|
||||
from scepter.modules.utils.sampler import MultiFoldDistributedSampler,
|
||||
EvalDistributedSampler, MultiLevelBatchSampler, MixtureOfSamplers
|
||||
```
|
||||
<hr/>
|
||||
@@ -832,17 +832,17 @@ from scepter.utils.sampler import MultiFoldDistributedSampler,
|
||||
|
||||
迭代器,每迭代一次得到一个样本的index
|
||||
|
||||
## 12. 探针器sdk(scepter.utils.probe)
|
||||
## 12. 探针器sdk(scepter.modules.utils.probe)
|
||||
用于探针各个组件的变量统计
|
||||
|
||||
### 基础用法
|
||||
|
||||
```python
|
||||
import numpy as np
|
||||
from scepter.model.base_model import BaseModel
|
||||
from scepter.utils.config import Config
|
||||
from scepter.utils.file_system import FS
|
||||
from scepter.utils.probe import ProbeData
|
||||
from scepter.modules.model.base_model import BaseModel
|
||||
from scepter.modules.utils.config import Config
|
||||
from scepter.modules.utils.file_system import FS
|
||||
from scepter.modules.utils.probe import ProbeData
|
||||
|
||||
|
||||
class TestModel(BaseModel):
|
||||
@@ -914,7 +914,7 @@ data = {
|
||||
_model(data)
|
||||
probe = _model.probe_data()
|
||||
for key in probe:
|
||||
print(key, probe[key].to_log(prefix=f"xxx/dev_easytorch/{key}"))
|
||||
print(key, probe[key].to_log(prefix=f"xxx/{key}"))
|
||||
```
|
||||
<hr/>
|
||||
|
||||
|
||||
@@ -6,4 +6,5 @@ dependencies:
|
||||
- pip>=20.3
|
||||
- numpy>=1.23.1
|
||||
- pip:
|
||||
- -r requirements/recommended.txt
|
||||
- -r requirements.txt
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
@@ -0,0 +1,3 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
from .classifier_dataset import ImageClassifyExampleDataset
|
||||
@@ -0,0 +1,260 @@
|
||||
ENV:
|
||||
USE_PL: False
|
||||
# SET GLOBAL SYSTEM
|
||||
SOLVER:
|
||||
# NAME DESCRIPTION: TYPE: default: 'TrainValSolver'
|
||||
NAME: TrainValSolver
|
||||
# RESUME_FROM DESCRIPTION: Resume from some state of training! TYPE: str default: ''
|
||||
RESUME_FROM:
|
||||
# MAX_EPOCHS DESCRIPTION: Max epochs for training. TYPE: int default: 10
|
||||
MAX_EPOCHS: 200
|
||||
# NUM_FOLDS DESCRIPTION: Num folds for training. TYPE: int default: 0
|
||||
NUM_FOLDS: 1
|
||||
# WORK_DIR DESCRIPTION: Save dir of the training log or model. TYPE: str default: ''
|
||||
WORK_DIR: ./exp12/
|
||||
LOG_FILE: std_log.txt
|
||||
# EVAL_INTERVAL DESCRIPTION: Eval the model interval. TYPE: int default: 1
|
||||
EVAL_INTERVAL: 1
|
||||
ACCU_STEP: 1
|
||||
# DO_FINAL_EVAL DESCRIPTION: If do final evaluation or not. TYPE: bool default: False
|
||||
DO_FINAL_EVAL: True
|
||||
# SAVE_EVAL_DATA DESCRIPTION: If save the evaluation data or not. TYPE: bool default: False
|
||||
SAVE_EVAL_DATA: True
|
||||
# EXTRA_KEYS DESCRIPTION: The extra keys for metric. TYPE: list default: []
|
||||
EXTRA_KEYS: []
|
||||
# TRAIN_DATA DESCRIPTION: Train data config. TYPE: default: ''
|
||||
TRAIN_DATA:
|
||||
# NAME DESCRIPTION: TYPE: default: 'ImageClassifyPublicDataset'
|
||||
NAME: ImageClassifyExampleDataset
|
||||
# DATASET DESCRIPTION: the public dataset name TYPE: str default: 'cifar10'
|
||||
DATASET: cifar10
|
||||
# DATA_ROOT DESCRIPTION: the download data save path TYPE: str default: ''
|
||||
DATA_ROOT: cifar10
|
||||
# MODE DESCRIPTION: test TYPE: str default: test
|
||||
MODE: train
|
||||
# PIN_MEMORY DESCRIPTION: pin_memory for data loader TYPE: bool default: False
|
||||
PIN_MEMORY: True
|
||||
# BATCH_SIZE DESCRIPTION: batch size for data TYPE: int default: 4
|
||||
BATCH_SIZE: 96
|
||||
# NUM_WORKERS DESCRIPTION: num workers for fetching data! TYPE: int default: 1
|
||||
NUM_WORKERS: 4
|
||||
# TRANSFORMS DESCRIPTION: TYPE: default:
|
||||
TRANSFORMS:
|
||||
# - DESCRIPTION: TYPE: default:
|
||||
- # NAME DESCRIPTION: TYPE: default: 'RandomResizedCrop'
|
||||
NAME: RandomResizedCrop
|
||||
SIZE: 32
|
||||
# RATIO DESCRIPTION: ratio TYPE: list default: [0.75, 1.3333333333333333]
|
||||
RATIO: [0.75, 1.33]
|
||||
# SCALE DESCRIPTION: scale TYPE: list default: [0.08, 1.0]
|
||||
SCALE: [0.8, 1.0]
|
||||
# INTERPOLATION DESCRIPTION: interpolation TYPE: str default: 'blilinear'
|
||||
INTERPOLATION: bilinear
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- # NAME DESCRIPTION: TYPE: default: 'RandomHorizontalFlip'
|
||||
NAME: RandomHorizontalFlip
|
||||
# P DESCRIPTION: P TYPE: float default: 0.5
|
||||
P: 0.5
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- # NAME DESCRIPTION: TYPE: default: 'ImageToTensor'
|
||||
NAME: ImageToTensor
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- # NAME DESCRIPTION: TYPE: default: 'Normalize'
|
||||
NAME: Normalize
|
||||
# MEAN DESCRIPTION: mean TYPE: list default: []
|
||||
MEAN: [0.4914, 0.4822, 0.4465]
|
||||
# STD DESCRIPTION: std TYPE: list default: []
|
||||
STD: [0.2023, 0.1994, 0.2010]
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- NAME: ToTensor
|
||||
# KEYS DESCRIPTION: keys TYPE: list default: []
|
||||
KEYS: ["img", "label"]
|
||||
- # NAME DESCRIPTION: TYPE: default: 'Select'
|
||||
NAME: Select
|
||||
# KEYS DESCRIPTION: keys TYPE: list default: []
|
||||
KEYS: ["img", "label"]
|
||||
# META_KEYS DESCRIPTION: meta keys TYPE: list default: []
|
||||
META_KEYS: []
|
||||
# EVAL_DATA DESCRIPTION: Eval data config. TYPE: default: ''
|
||||
EVAL_DATA:
|
||||
# NAME DESCRIPTION: TYPE: default: 'ImageClassifyPublicDataset'
|
||||
NAME: ImageClassifyPublicDataset
|
||||
# DATASET DESCRIPTION: the public dataset name TYPE: str default: 'cifar10'
|
||||
DATASET: cifar10
|
||||
# DATA_ROOT DESCRIPTION: the download data save path TYPE: str default: ''
|
||||
DATA_ROOT: ./local_data/cifar10
|
||||
# MODE DESCRIPTION: test TYPE: str default: test
|
||||
MODE: test
|
||||
# PIN_MEMORY DESCRIPTION: pin_memory for data loader TYPE: bool default: False
|
||||
PIN_MEMORY: True
|
||||
# BATCH_SIZE DESCRIPTION: batch size for data TYPE: int default: 4
|
||||
BATCH_SIZE: 96
|
||||
# NUM_WORKERS DESCRIPTION: num workers for fetching data! TYPE: int default: 1
|
||||
NUM_WORKERS: 4
|
||||
# TRANSFORMS DESCRIPTION: TYPE: default:
|
||||
TRANSFORMS:
|
||||
# - DESCRIPTION: TYPE: default:
|
||||
- # NAME DESCRIPTION: TYPE: default: 'Resize'
|
||||
NAME: Resize
|
||||
SIZE: 32
|
||||
# INTERPOLATION DESCRIPTION: interpolation TYPE: str default: 'blilinear'
|
||||
INTERPOLATION: bilinear
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- # NAME DESCRIPTION: TYPE: default: 'ImageToTensor'
|
||||
NAME: ImageToTensor
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- # NAME DESCRIPTION: TYPE: default: 'Normalize'
|
||||
NAME: Normalize
|
||||
# MEAN DESCRIPTION: mean TYPE: list default: []
|
||||
MEAN: [0.4914, 0.4822, 0.4465]
|
||||
# STD DESCRIPTION: std TYPE: list default: []
|
||||
STD: [0.2023, 0.1994, 0.2010]
|
||||
# INPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
INPUT_KEY: img
|
||||
# OUTPUT_KEY DESCRIPTION: input key TYPE: str default: 'img'
|
||||
OUTPUT_KEY: img
|
||||
# BACKEND DESCRIPTION: backend, choose from pillow, cv2, torchvision TYPE: str default: 'pillow'
|
||||
BACKEND: pillow
|
||||
- NAME: ToTensor
|
||||
# KEYS DESCRIPTION: keys TYPE: list default: []
|
||||
KEYS: ["img", "label"]
|
||||
- # NAME DESCRIPTION: TYPE: default: 'Select'
|
||||
NAME: Select
|
||||
# KEYS DESCRIPTION: keys TYPE: list default: []
|
||||
KEYS: ["img", "label"]
|
||||
# META_KEYS DESCRIPTION: meta keys TYPE: list default: []
|
||||
META_KEYS: []
|
||||
# TRAIN_HOOKS DESCRIPTION: TYPE: default: ''
|
||||
TRAIN_HOOKS:
|
||||
- # NAME DESCRIPTION: TYPE: default: 'LogHook'
|
||||
NAME: LogHook
|
||||
# LOG_INTERVAL DESCRIPTION: the interval for log print! TYPE: int default: 10
|
||||
LOG_INTERVAL: 10
|
||||
# EVAL_HOOKS DESCRIPTION: TYPE: default: ''
|
||||
EVAL_HOOKS:
|
||||
- # NAME DESCRIPTION: TYPE: default: 'LogHook'
|
||||
NAME: LogHook
|
||||
# LOG_INTERVAL DESCRIPTION: the interval for log print! TYPE: int default: 10
|
||||
LOG_INTERVAL: 10
|
||||
# TEST_HOOKS DESCRIPTION: TYPE: default: ''
|
||||
MODEL:
|
||||
# NAME DESCRIPTION: TYPE: default: 'Classifier'
|
||||
NAME: Classifier
|
||||
# ACT_NAME DESCRIPTION: the activation function for logits, select from [softmax, sigmoid]! TYPE: str default: 'softmax'
|
||||
ACT_NAME: softmax
|
||||
# FREEZE_BN DESCRIPTION: if freeze bn of not TYPE: bool default: False
|
||||
FREEZE_BN: False
|
||||
# BACKBONE DESCRIPTION: TYPE: default: ''
|
||||
BACKBONE:
|
||||
# NAME DESCRIPTION: TYPE: default: 'ResNet'
|
||||
NAME: ResNet
|
||||
# DEPTH DESCRIPTION: the depth of network for resnet! TYPE: int default: 18
|
||||
DEPTH: 18
|
||||
# PRETRAINED DESCRIPTION: if load the official pretrained model or not. TYPE: bool default: False
|
||||
PRETRAINED: false
|
||||
#
|
||||
KERNEL_SIZE: 3
|
||||
# USE_RELU DESCRIPTION: use relu or not! TYPE: bool default: True
|
||||
USE_RELU: True
|
||||
# USE_MAXPOOL DESCRIPTION: use maxpool or not! TYPE: bool default: True
|
||||
USE_MAXPOOL: false
|
||||
# FIRST_CONV_STRIDE DESCRIPTION: first conv stride 1 or 2! TYPE: int default: 1
|
||||
FIRST_CONV_STRIDE: 1
|
||||
# FIRST_MAX_POOL_STRIDE DESCRIPTION: first max pool stride 1 or 2! TYPE: int default: 1
|
||||
FIRST_MAX_POOL_STRIDE: 1
|
||||
# NECK DESCRIPTION: TYPE: default: ''
|
||||
NECK:
|
||||
# NAME DESCRIPTION: TYPE: default: 'GlobalAveragePooling'
|
||||
NAME: GlobalAveragePooling
|
||||
# DIM DESCRIPTION: GlobalAveragePooling dim! TYPE: int default: 2
|
||||
DIM: 2
|
||||
# HEAD DESCRIPTION: TYPE: default: ''
|
||||
HEAD:
|
||||
# NAME DESCRIPTION: TYPE: default: 'ClassifierHead'
|
||||
NAME: ClassifierHead
|
||||
# DIM DESCRIPTION: representation dim! TYPE: int default: 512
|
||||
DIM: 512
|
||||
# NUM_CLASSES DESCRIPTION: number of classes. TYPE: int default: 10
|
||||
NUM_CLASSES: 10
|
||||
# DROPOUT_RATE DESCRIPTION: dropout rate, default 0. TYPE: float default: 0.0
|
||||
DROPOUT_RATE: 0.0
|
||||
METRIC:
|
||||
# NAME DESCRIPTION: TYPE: default: 'AccuracyMetric'
|
||||
NAME: AccuracyMetric
|
||||
# TOPK DESCRIPTION: topk accuracy! TYPE: int default: 1
|
||||
TOPK: 1
|
||||
# LOSS DESCRIPTION: TYPE: default: ''
|
||||
LOSS:
|
||||
# NAME DESCRIPTION: TYPE: default: 'CrossEntropy'
|
||||
NAME: CrossEntropy
|
||||
# REDUCE DESCRIPTION: reduce is False, returns a loss per batch element instead and ignores :attr: size_average. Default: True TYPE: NoneType default: None
|
||||
# REDUCE: None
|
||||
# SIZE_AVERAGE DESCRIPTION: Deprecated (see :attr: reduction). By default,the losses are averaged over each loss element in the batch. Note that forsome losses, there are multiple elements per sample. If the field :attr: size_averageis set to False, the losses are instead summed for each minibatch. Ignoredwhen :attr: reduce is False. Default: True TYPE: NoneType default: None
|
||||
# SIZE_AVERAGE: None
|
||||
# IGNORE_INDEX DESCRIPTION: Specifies a target value that is ignoredand does not contribute to the input gradient. When :attr: size_average isTrue, the loss is averaged over non-ignored targets. Note that:attr: ignore_index is only applicable when the target contains class indices. TYPE: int default: -100
|
||||
# IGNORE_INDEX: -100
|
||||
# REDUCTION DESCRIPTION: Specifies the reduction to apply to the output:'none' | 'mean' | 'sum'. 'none': no reduction willbe applied, 'mean': the weighted mean of the output is taken,'sum': the output will be summed. Note: :attr: size_averageand :attr:`reduce` are in the process of being deprecated, and inthe meantime, specifying either of those two args will override:attr:`reduction`. Default: 'mean' TYPE: str default: 'mean'
|
||||
# REDUCTION: mean
|
||||
# LABEL_SMOOTHING DESCRIPTION: A float in [0.0, 1.0]. Specifies the amountof smoothing when computing the loss, where 0.0 means no smoothing. TYPE: float default: 0.0
|
||||
# LABEL_SMOOTHING: 0.0
|
||||
# OPTIMIZER DESCRIPTION: TYPE: default: ''
|
||||
OPTIMIZER:
|
||||
# NAME DESCRIPTION: TYPE: default: 'SGD'
|
||||
NAME: SGD
|
||||
# LEARNING_RATE DESCRIPTION: the initial learning rate! TYPE: float default: 0.1
|
||||
LEARNING_RATE: 0.01
|
||||
# MOMENTUM DESCRIPTION: the momentum! TYPE: int default: 0
|
||||
MOMENTUM: 0.9
|
||||
# DAMPENING DESCRIPTION: the dampening! TYPE: int default: 0
|
||||
DAMPENING: 0
|
||||
# WEIGHT_DECAY DESCRIPTION: the weight decay! TYPE: int default: 0
|
||||
WEIGHT_DECAY: 5e-4
|
||||
# NESTEROV DESCRIPTION: the nesterov! TYPE: bool default: False
|
||||
NESTEROV: False
|
||||
# LR_SCHEDULER DESCRIPTION: TYPE: default: ''
|
||||
LR_SCHEDULER:
|
||||
# NAME DESCRIPTION: TYPE: default: 'CosineAnnealingLR'
|
||||
NAME: CosineAnnealingLR
|
||||
# T_MAX DESCRIPTION: the T max! TYPE: float default: 1.0
|
||||
T_MAX: 200.0
|
||||
# ETA_MIN DESCRIPTION: the eta min! TYPE: int default: 0
|
||||
ETA_MIN: 0
|
||||
# LAST_EPOCH DESCRIPTION: the last epoch! TYPE: int default: -1
|
||||
LAST_EPOCH: -1
|
||||
# METRICS DESCRIPTION: TYPE: default: ''
|
||||
METRICS:
|
||||
- # NAME DESCRIPTION: TYPE: default: 'AccuracyMetric'
|
||||
NAME: AccuracyMetric
|
||||
# TOPK DESCRIPTION: topk accuracy! TYPE: int default: 1
|
||||
TOPK: 1
|
||||
KEYS: ["logits", "label"]
|
||||
@@ -0,0 +1,79 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import numpy as np
|
||||
import torchvision
|
||||
from scepter.modules.data.dataset.base_dataset import BaseDataset
|
||||
from scepter.modules.data.dataset.registry import DATASETS
|
||||
from scepter.modules.utils.config import dict_to_yaml
|
||||
|
||||
|
||||
@DATASETS.register_class()
|
||||
class ImageClassifyExampleDataset(BaseDataset):
|
||||
"""
|
||||
Dataset for image classification wrapper
|
||||
|
||||
Args:
|
||||
json_path (str): json file which contains all instances, should be a list of dict
|
||||
which contains img_path and gt_label
|
||||
image_dir (str or None): image directory, if None, img_path in json_path will be considered as absolute path
|
||||
classes (list[str] or None): image class description
|
||||
"""
|
||||
para_dict = {
|
||||
'DATASET': {
|
||||
'value': 'cifar10',
|
||||
'description': 'the public dataset name'
|
||||
},
|
||||
'DATA_ROOT': {
|
||||
'value': '',
|
||||
'description': 'the download data save path'
|
||||
}
|
||||
}
|
||||
|
||||
para_dict.update(BaseDataset.para_dict)
|
||||
|
||||
def __init__(self, cfg, logger=None):
|
||||
|
||||
super(ImageClassifyExampleDataset, self).__init__(cfg, logger=logger)
|
||||
|
||||
self.dataset_name = cfg.DATASET
|
||||
self.data_root = cfg.DATA_ROOT
|
||||
self.phase = cfg.MODE
|
||||
if self.dataset_name == 'cifar10':
|
||||
self.dataset = torchvision.datasets.CIFAR10(
|
||||
root=self.data_root,
|
||||
train=self.phase == 'train',
|
||||
download=True)
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self.dataset)
|
||||
|
||||
def _get(self, index: int):
|
||||
img, target = self.dataset.__getitem__(index)
|
||||
ret = {
|
||||
'meta': {},
|
||||
'label': np.asarray(target, dtype=np.int64),
|
||||
'img': img
|
||||
}
|
||||
return ret
|
||||
|
||||
def worker_init_fn(self, worker_id, num_workers=1):
|
||||
super(ImageClassifyExampleDataset,
|
||||
self).worker_init_fn(worker_id, num_workers=num_workers)
|
||||
|
||||
@staticmethod
|
||||
def get_config_template():
|
||||
'''
|
||||
{ "ENV" :
|
||||
{ "description" : "",
|
||||
"A" : {
|
||||
"value": 1.0,
|
||||
"description": ""
|
||||
}
|
||||
}
|
||||
}
|
||||
:return:
|
||||
'''
|
||||
return dict_to_yaml('modename_DATA',
|
||||
__class__.__name__,
|
||||
ImageClassifyExampleDataset.para_dict,
|
||||
set_name=True)
|
||||
@@ -0,0 +1,7 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
from scepter.tools.run_train import run
|
||||
|
||||
if __name__ == '__main__':
|
||||
run()
|
||||
@@ -8,156 +8,83 @@
|
||||
<a href="https://github.com/modelscope/scepter/"><img src="https://img.shields.io/badge/scepter-Build from source-6FEBB9.svg"></a>
|
||||
</p>
|
||||
|
||||
## 📖 Table of Contents
|
||||
- [Introduction](#-introduction)
|
||||
- [News](#-news)
|
||||
- [Installation](#-Installation)
|
||||
- [Getting Started](#-getting-started)
|
||||
- [SCEPTER Studio](#-scepter-studio)
|
||||
- [Features](#-features)
|
||||
- [Learn More](#-learn-more)
|
||||
- [License](#license)
|
||||
🪄SCEPTER is an open-source code repository dedicated to generative training, fine-tuning, and inference, encompassing a suite of downstream tasks such as image generation, transfer, editing.
|
||||
SCEPTER integrates popular community-driven implementations as well as proprietary methods by Tongyi Lab of Alibaba Group, offering a comprehensive toolkit for researchers and practitioners in the field of AIGC. This versatile library is designed to facilitate innovation and accelerate development in the rapidly evolving domain of generative models.
|
||||
|
||||
## 📝 Introduction
|
||||
SCEPTER offers 3 core components:
|
||||
- [Generative training and inference framework](#tutorials)
|
||||
- [Easy implementation of popular approaches](#currently-supported-approaches)
|
||||
- [Interactive user interface: SCEPTER Studio](#launch)
|
||||
|
||||
SCEPTER is an open-source code repository dedicated to generative training, fine-tuning, and inference, encompassing a suite of downstream tasks such as image generation, transfer, editing. It integrates popular community-driven implementations as well as proprietary methods by Tongyi Lab of Alibaba Group, offering a comprehensive toolkit for researchers and practitioners in the field of AIGC. This versatile library is designed to facilitate innovation and accelerate development in the rapidly evolving domain of generative models.
|
||||
|
||||
Main Feature:
|
||||
|
||||
- Task:
|
||||
- Text-to-image generation
|
||||
- Controllable image synthesis
|
||||
- Image editing (TODO)
|
||||
- Training / Inference:
|
||||
- Distribute: DDP / FSDP / FairScale / Xformers
|
||||
- File system: Local / Http / OSS / Modelscope
|
||||
- Deploy:
|
||||
- Data management
|
||||
- Training
|
||||
- Inference
|
||||
|
||||
Currently supported approches (and counting):
|
||||
|
||||
1. SD Series: [Stable Diffusion v1.5](https://huggingface.co/runwayml/stable-diffusion-v1-5) / [Stable Diffusion v2.1](https://huggingface.co/runwayml/stable-diffusion-v1-5) / [Stable Diffusion XL](https://huggingface.co/stabilityai/stable-diffusion-xl-base-1.0)
|
||||
2. SCEdit: [SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing](https://arxiv.org/abs/2312.11392) [](https://arxiv.org/abs/2312.11392) [](https://scedit.github.io/)
|
||||
3. Res-Tuning(TODO): [Res-Tuning: A Flexible and Efficient Tuning Paradigm via Unbinding Tuner from Backbone](https://arxiv.org/abs/2310.19859) [](https://arxiv.org/abs/2310.19859) [](https://res-tuning.github.io/)
|
||||
|
||||
## 🎉 News
|
||||
- [2024.09]: We introduce **ACE**, an **A**ll-round **C**reator and **E**ditor adept at executing a diverse array of image editing tasks tailored to your specifications. Built upon the cutting-edge Diffusion Transformer architecture, ACE has been extensively trained on a comprehensive dataset to seamlessly interpret and execute any natural language instruction. For further information, please consult the [project page]().
|
||||
- [2024.07]: Support the inference and training of open-source generative models based on the [DiT](https://arxiv.org/abs/2212.09748) architecture, such as [SD3](https://arxiv.org/pdf/2403.03206) and [PixArt](https://arxiv.org/abs/2310.00426).
|
||||
- [2024.05]: Introducing SCEPTER v1, supporting customized image edit tasks! Simply provide 10 image pairs, SCEPTER will tune an edit tuner for your own Image-to-Image tasks, like `Clay Style`, `De-Text`, `Segmentation`, etc.
|
||||
- [2024.04]: New [StyleBooth](https://ali-vilab.github.io/stylebooth-page/) demo on SCEPTER Studio for`Text-Based Style Editing`.
|
||||
- [2024.03]: We optimize the training UI and checkpoint management. New [LAR-Gen](https://arxiv.org/abs/2403.19534) model has been added on SCEPTER Studio, supporting `zoom-out`, `virtual try on`, `inpainting`.
|
||||
- [2024.02]: We release new SCEdit controllable image synthesis models for SD v2.1 and SD XL. Multiple strategies applied to accelerate inference time for SCEPTER Studio.
|
||||
- [2024.01]: We release **SCEPTER Studio**, an integrated toolkit for data management, model training and inference based on [Gradio](https://www.gradio.app/).
|
||||
- [2024.01]: [SCEdit](https://arxiv.org/abs/2312.11392) support controllable image synthesis for training and inference.
|
||||
- [2023.12]: We propose [SCEdit](https://arxiv.org/abs/2312.11392), an efficient and controllable generation framework.
|
||||
- [2023.12]: We release [🪄SCEPTER](https://github.com/modelscope/scepter/) library.
|
||||
|
||||
|
||||
## 🖼 Gallery for Recent Works
|
||||
|
||||
### <img src="asset/images/ace/logo.png" height=20> <img src="asset/images/ace/text.png" height=20>
|
||||
|
||||
ACE is a unified foundational model framework that supports a wide range of visual generation tasks. By defining CU for unifying multi-modal inputs across different tasks and incorporating long-
|
||||
context CU, we introduce historical contextual information into visual generation tasks, paving
|
||||
the way for ChatGPT-like dialog systems in visual generation.
|
||||
|
||||
<a href="https://ali-vilab.github.io/ace-page/">
|
||||
<img src="asset/images/ace/teaser_dy.gif" width=1024>
|
||||
</a>
|
||||
|
||||
|
||||
## 🛠️ Installation
|
||||
|
||||
- Create new environment
|
||||
- Create new environment with `conda` command:
|
||||
|
||||
```shell
|
||||
conda env create -f environment.yaml
|
||||
conda activate scepter
|
||||
```
|
||||
|
||||
- Install SCEPTER by the `pip` command:
|
||||
- Install with `pip` command:
|
||||
|
||||
We recommend installing the specific version of PyTorch and accelerate toolbox [xFormers](https://pypi.org/project/xformers/). You can install these recommended version by pip:
|
||||
```shell
|
||||
pip install -r requirements/recommended.txt
|
||||
pip install scepter
|
||||
```
|
||||
- PS: We recommend installing PyTorch follwing [official documentation](https://pytorch.org/get-started/locally/)
|
||||
|
||||
## 🚀 Getting Started
|
||||
## 🧩 Generative Framework
|
||||
|
||||
### Dataset
|
||||
### Tutorials
|
||||
|
||||
#### Modelscope Format
|
||||
| Documentation | Key Features |
|
||||
|:---------------------------------------------------|:----------------------------------|
|
||||
| [Train](docs/en/tutorials/train.md) | DDP / FSDP / FairScale / Xformers |
|
||||
| [Inference](docs/en/tutorials/inference.md) | Dynamic load/unload |
|
||||
| [Dataset Management](docs/en/tutorials/dataset.md) | Local / Http / OSS / Modelscope |
|
||||
|
||||
We use a [custom-stylized dataset](https://modelscope.cn/datasets/damo/style_custom_dataset/summary), which included classes 3D, anime, flat illustration, oil painting, sketch, and watercolor, each with 30 image-text pairs.
|
||||
|
||||
```python
|
||||
# pip install modelscope
|
||||
from modelscope.msdatasets import MsDataset
|
||||
ms_train_dataset = MsDataset.load('style_custom_dataset', namespace='damo', subset_name='3D', split='train_short')
|
||||
print(next(iter(ms_train_dataset)))
|
||||
```
|
||||
## 📝 Popular Approaches
|
||||
|
||||
#### CSV Format
|
||||
### Currently supported approaches
|
||||
|
||||
For the data format used by SCEPTER Studio, please refer to [3D_example_csv.zip](https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=datasets/3D_example_csv.zip).
|
||||
|
||||
#### TXT Format
|
||||
|
||||
To facilitate starting training in command-line mode, you can use a dataset in text format, please refer to [3D_example_txt.zip](https://modelscope.cn/api/v1/models/damo/scepter/repo?Revision=master&FilePath=datasets/3D_example_txt.zip)
|
||||
|
||||
```shell
|
||||
mkdir -p cache/dataset/ && wget 'https://modelscope.cn/api/v1/models/damo/scepter_scedit/repo?Revision=master&FilePath=dataset/3D_example_txt.zip' -O cache/dataset/3D_example_txt.zip && unzip cache/dataset/3D_example_txt.zip -d cache/dataset/ && rm cache/dataset/3D_example_txt.zip
|
||||
```
|
||||
|
||||
### Training
|
||||
|
||||
We provide a framework for training and inference, so the script below is just for illustration purposes. To achieve better results, you can modify the corresponding parameters as needed.
|
||||
|
||||
#### Text-to-Image Generation
|
||||
|
||||
- SCEdit
|
||||
```python
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i.yaml # SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sd21_768_sce_t2i.yaml # SD v2.1
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i.yaml # SD XL
|
||||
```
|
||||
|
||||
- Existing Tuning Strategies
|
||||
```python
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_1.5_512.yaml # fully-tuning on SD v1.5
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/examples/generation/stable_diffusion_2.1_768_lora.yaml # lora-tuning on SD v2.1
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```python
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/t2i/sdxl_1024_sce_t2i_datatxt.yaml
|
||||
```
|
||||
|
||||
#### Controllable Image Synthesis
|
||||
|
||||
- SCEdit
|
||||
|
||||
The YAML configuration can be modified to combine different base models and conditions. The following is provided as an example.
|
||||
```python
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd15_512_sce_ctr_hed.yaml # SD v1.5 + hed
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml # SD v2.1 + canny
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml # SD v2.1 + pose
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_depth.yaml # SD XL + depth
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color.yaml # SD XL + color
|
||||
```
|
||||
|
||||
- Data Text Format
|
||||
```python
|
||||
# Download the 3D_example_txt.zip as previously mentioned
|
||||
python scepter/tools/run_train.py --cfg scepter/methods/scedit/ctr/sdxl_1024_sce_ctr_color_datatxt.yaml
|
||||
```
|
||||
|
||||
### Inference
|
||||
|
||||
#### Base Model Inference
|
||||
|
||||
```python
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_1.5_512.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD v1.5
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_2.1_768.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD v2.1
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/examples/generation/stable_diffusion_xl_1024.yaml --prompt 'a cute dog' --save_folder 'inference' # generation on SD XL
|
||||
```
|
||||
|
||||
#### Fine-tuned Model Inference
|
||||
|
||||
```python
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/t2i/sd15_512_sce_t2i_swift.yaml --pretrained_model 'cache/save_data/sd15_512_sce_t2i_swift/checkpoints/ldm_step-100.pth' --prompt 'A close up of a small rabbit wearing a hat and scarf' --save_folder 'trained_test_prompt_rabbit'
|
||||
```
|
||||
|
||||
#### Controllable Image Synthesis Inference
|
||||
|
||||
- SCEdit
|
||||
```python
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_canny.yaml --num_samples 1 --prompt 'a single flower is shown in front of a tree' --save_folder 'test_flower_canny' --image_size 768 --task control --image 'asset/images/flower.jpg' --control_mode canny --pretrained_model ms://damo/scepter_scedit@controllable_model/SD2.1/canny_control/0_SwiftSCETuning/pytorch_model.bin # canny
|
||||
python scepter/tools/run_inference.py --cfg scepter/methods/scedit/ctr/sd21_768_sce_ctr_pose.yaml --num_samples 1 --prompt 'super mario' --save_folder 'test_mario_pose' --image_size 768 --task control --image 'asset/images/pose_source.png' --control_mode source --pretrained_model ms://damo/scepter_scedit@controllable_model/SD2.1/pose_control/0_SwiftSCETuning/pytorch_model.bin # pose
|
||||
```
|
||||
| Tasks | Methods | Links |
|
||||
|:----------------------------:|:--------------------------------------------:|:------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Text-to-image generation | SD v1.5 | [](https://huggingface.co/runwayml/stable-diffusion-v1-5) |
|
||||
| Text-to-image generation | SD v2.1 | [](https://huggingface.co/runwayml/stable-diffusion-v1-5) |
|
||||
| Text-to-image generation | SD-XL | [](https://huggingface.co/stabilityai/stable-diffusion-xl-base-1.0) |
|
||||
| Efficient Tuning | LoRA | [](https://arxiv.org/abs/2106.09685) |
|
||||
| Efficient Tuning | Res-Tuning(NeurIPS23) | [](https://arxiv.org/abs/2310.19859) [](https://res-tuning.github.io/) |
|
||||
| Controllable image synthesis | [🌟SCEdit(CVPR24)](docs/en/tasks/scedit.md) | [](https://arxiv.org/abs/2312.11392) [](https://scedit.github.io/) |
|
||||
| Image editing | [🌟LAR-Gen](docs/en/tasks/largen.md) | [](https://arxiv.org/abs/2403.19534) [](https://ali-vilab.github.io/largen-page/) |
|
||||
| Image editing | [🌟StyleBooth](docs/en/tasks/stylebooth.md) | [](https://arxiv.org/abs/2404.12154) [](https://ali-vilab.github.io/stylebooth-page/) |
|
||||
|
||||
|
||||
## 🖥️ SCEPTER Studio
|
||||
@@ -176,50 +103,24 @@ git clone https://github.com/modelscope/scepter.git
|
||||
PYTHONPATH=. python scepter/tools/webui.py --cfg scepter/methods/studio/scepter_ui.yaml
|
||||
```
|
||||
|
||||
The startup of **SCEPTER Studio** eliminates the need for manual downloading and organizing of models; it will automatically load the corresponding models and store them in a local directory.
|
||||
Depending on the network and hardware situation, the initial startup usually requires 15-60 minutes, primarily involving the download and processing of SDv1.5, SDv2.1, and SDXL models.
|
||||
The startup of **SCEPTER Studio** eliminates the need for manual downloading and organizing of models; it will automatically load the corresponding models and store them in a local directory.
|
||||
Depending on the network and hardware situation, the initial startup usually requires 15-60 minutes, primarily involving the download and processing of SDv1.5, SDv2.1, and SDXL models.
|
||||
Therefore, subsequent startups will become much faster (about one minute) as downloading is no longer required.
|
||||
|
||||
### Usage Demo
|
||||
|
||||
### Modelscope Studio
|
||||
| [Image Editing](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fimage_editing_20240419.webm) | [Training](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Ftraining_20240419.webm) | [Model Sharing](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_sharing_20240419.webm) | [Model Inference](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_inference_20240419.webm) | [Data Management](https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fdata_management_20240419.webm) |
|
||||
|:----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:--------------------------------------------:|
|
||||
| <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fimage_editing_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Ftraining_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_sharing_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fmodel_inference_20240419.webm" width="240" controls></video> | <video src="https://www.modelscope.cn/api/v1/models/iic/scepter/repo?Revision=master&FilePath=assets%2Fscepter_studio%2Fdata_management_20240419.webm" width="240" controls></video> |
|
||||
|
||||
We deploy a work studio on Modelscope that includes only the inference tab, please refer to [ms_scepter_studio](https://www.modelscope.cn/studios/damo/scepter_studio/summary)
|
||||
### Modelscope Studio & Huggingface Space
|
||||
|
||||
## ✨ Features
|
||||
|
||||
### Text-to-Image Generation
|
||||
|
||||
| **Model** | **SCEdit** | **Full** | **LoRA** |
|
||||
|:---------:|:----------:|:--------:|:--------:|
|
||||
| SD 1.5 | 🪄 | ✅ | ✅ |
|
||||
| SD 2.1 | 🪄 | ✅ | ✅ |
|
||||
| SD XL | 🪄 | ✅ | ✅ |
|
||||
|
||||
### Controllable Image Synthesis
|
||||
- SCEdit
|
||||
|
||||
| **Model** | **Canny** | **HED** | **Depth** | **Pose** | **Color** |
|
||||
|:---------:|:---------:|:-------:|:---------:|:--------:|:---------:|
|
||||
| SD 1.5 | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| SD 2.1 | 🪄 | ✅ | ✅ | 🪄 | 🪄 |
|
||||
| SD XL | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
|
||||
### Model URL
|
||||
|
||||
- ✅ indicates support for both training and inference.
|
||||
- 🪄 denotes that the model has been published.
|
||||
- More models will be released in the future.
|
||||
|
||||
| Model | URL |
|
||||
|--------|-------------------------------------------------------------------------------------|
|
||||
| SCEdit | [ModelCard](https://modelscope.cn/models/damo/scepter_scedit/summary) |
|
||||
|
||||
PS: Scripts running within the SCEPTER framework will automatically fetch and load models based on the required dependency files, eliminating the need for manual downloads.
|
||||
We deploy a work studio on Modelscope that includes only the inference tab, please refer to [ms_scepter_studio](https://www.modelscope.cn/studios/iic/scepter_studio/summary) and [hf_scepter_studio](https://huggingface.co/spaces/modelscope/scepter_studio)
|
||||
|
||||
|
||||
## 🔍 Learn More
|
||||
|
||||
- [Alibaba TongYi Vision Intelligence Lab](https://github.com/damo-vilab)
|
||||
- [Alibaba TongYi Vision Intelligence Lab](https://github.com/ali-vilab)
|
||||
|
||||
Discover more about open-source projects on image generation, video generation, and editing tasks.
|
||||
|
||||
@@ -232,6 +133,21 @@ PS: Scripts running within the SCEPTER framework will automatically fetch and lo
|
||||
SWIFT (Scalable lightWeight Infrastructure for Fine-Tuning) is an extensible framwork designed to faciliate lightweight model fine-tuning and inference.
|
||||
|
||||
|
||||
## BibTeX
|
||||
If our work is useful for your research, please consider citing:
|
||||
```bibtex
|
||||
@misc{scepter,
|
||||
title = {SCEPTER, https://github.com/modelscope/scepter},
|
||||
author = {SCEPTER},
|
||||
year = {2023}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## License
|
||||
|
||||
This project is licensed under the [Apache License (Version 2.0)](https://github.com/modelscope/modelscope/blob/master/LICENSE).
|
||||
|
||||
|
||||
## Acknowledgement
|
||||
Thanks to [Stability-AI](https://github.com/Stability-AI), [SWIFT library](https://github.com/modelscope/swift/) and [Fooocus](https://github.com/lllyasviel/Fooocus) for their awesome work.
|
||||
|
||||
@@ -1,11 +1,16 @@
|
||||
albumentations
|
||||
beautifulsoup4
|
||||
bezier
|
||||
einops
|
||||
modelscope
|
||||
ms-swift>=1.5.2
|
||||
modelscope==1.14.0
|
||||
ms-swift>=2.0.1
|
||||
numpy
|
||||
open_clip_torch
|
||||
opencv-python
|
||||
opencv_transforms>=0.0.6
|
||||
oss2>=2.15.0
|
||||
pycocotools
|
||||
pyyaml>=5.3.1
|
||||
scikit-image
|
||||
torchsde
|
||||
transformers
|
||||
xformers>=0.0.21
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
git+https://github.com/cocodataset/panopticapi.git
|
||||
torch==2.0.1
|
||||
torchvision==0.15.2
|
||||
xformers==0.0.21
|
||||
@@ -1,2 +1,6 @@
|
||||
gradio>=3.47.1,<4.0.0
|
||||
bitsandbytes
|
||||
gradio
|
||||
imagehash
|
||||
psutil
|
||||
tiktoken
|
||||
transformers_stream_generator
|
||||
|
||||
@@ -0,0 +1,250 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/edit_512_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionEdit
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://iic/stylebooth@models/stylebooth-tb-5000-0.bin
|
||||
IGNORE_KEYS: [ ]
|
||||
CONCAT_NO_SCALE_FACTOR: True
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 8
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: #7.5
|
||||
image: 1.5
|
||||
text: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: cache/datasets/hed_pair
|
||||
MS_DATASET_NAMESPACE: ""
|
||||
MS_DATASET_SPLIT: "train"
|
||||
MS_DATASET_SUBNAME: ""
|
||||
PROMPT_PREFIX: ""
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFileList
|
||||
FILE_KEYS: ['img_path', 'src_path']
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'img', 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img', 'src' ]
|
||||
OUTPUT_KEY: [ 'image', 'condition_cat' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'condition_cat', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
|
||||
EVAL_DATA:
|
||||
NAME: Text2ImageDataset
|
||||
MODE: eval
|
||||
PROMPT_FILE:
|
||||
PROMPT_DATA: [ "Convert to an edge map#;#cache/datasets/hed_pair/images/src_001.jpeg" ]
|
||||
IMAGE_SIZE: [ 512, 512 ]
|
||||
FIELDS: [ "prompt", "src_path" ]
|
||||
DELIMITER: '#;#'
|
||||
PROMPT_PREFIX: ''
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFileList
|
||||
FILE_KEYS: [ 'src_path' ]
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 512, 512 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'src' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'src' ]
|
||||
OUTPUT_KEY: [ 'condition_cat' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'condition_cat', 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
SAVE_PROBE_PREFIX: 'image'
|
||||
@@ -11,7 +11,7 @@ SOLVER:
|
||||
# NUM_FOLDS DESCRIPTION: Num folds for training. TYPE: int default: 0
|
||||
NUM_FOLDS: 1
|
||||
# WORK_DIR DESCRIPTION: Save dir of the training log or model. TYPE: str default: ''
|
||||
WORK_DIR: ./exp12/
|
||||
WORK_DIR: ./cache/save_data/example/
|
||||
LOG_FILE: std_log.txt
|
||||
# EVAL_INTERVAL DESCRIPTION: Eval the model interval. TYPE: int default: 1
|
||||
EVAL_INTERVAL: 1
|
||||
@@ -102,7 +102,7 @@ SOLVER:
|
||||
# DATASET DESCRIPTION: the public dataset name TYPE: str default: 'cifar10'
|
||||
DATASET: cifar10
|
||||
# DATA_ROOT DESCRIPTION: the download data save path TYPE: str default: ''
|
||||
DATA_ROOT: ./local_data/cifar10
|
||||
DATA_ROOT: ./cache/cache_data/cifar10
|
||||
# MODE DESCRIPTION: test TYPE: str default: test
|
||||
MODE: test
|
||||
# PIN_MEMORY DESCRIPTION: pin_memory for data loader TYPE: bool default: False
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
#
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 1000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/dit_pixart_alpha_1024_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
FREEZE:
|
||||
#
|
||||
TUNER:
|
||||
- NAME: SwiftLoRA
|
||||
R: 128
|
||||
LORA_ALPHA: 128
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(.q|.k|.v|.o|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionPixart
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
DECODER_BIAS: 0.5
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "linear"
|
||||
"BETA_MIN": 0.0001
|
||||
"BETA_MAX": 0.02
|
||||
USE_EMA: False
|
||||
LOAD_REFINER: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: PixArt
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/PixArt-alpha@PixArt-XL-2-1024-MS.pth
|
||||
INPUT_SIZE: 128
|
||||
PATCH_SIZE: 2
|
||||
IN_CHANNELS: 4
|
||||
HIDDEN_SIZE: 1152
|
||||
DEPTH: 28
|
||||
NUM_HEADS: 16
|
||||
MLP_RATIO: 4.0
|
||||
CLASS_DROPOUT_PROB: 0.1
|
||||
PRED_SIGMA: True
|
||||
DROP_PATH: 0.0
|
||||
WINDOW_DIZE: 0
|
||||
USE_REL_POS: False
|
||||
CAPTION_CHANNELS: 4096
|
||||
LEWEI_SCALE: 2
|
||||
MODEL_MAX_LENGTH: 120
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-2-base@512-base-ema.safetensors
|
||||
EMBED_DIM: 4
|
||||
IGNORE_KEYS: [ ]
|
||||
BATCH_SIZE: 1
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: T5EmbedderHF
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/PixArt-alpha@t5-v1_1-xxl/
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/PixArt-alpha@t5-v1_1-xxl/
|
||||
LENGTH: 120
|
||||
CLEAN: heavy
|
||||
USE_GRAD: False
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 20
|
||||
SEED: 2024
|
||||
GUIDE_SCALE: 4.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 1024, 1024 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 1024, 1024 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: Text2ImageDataset
|
||||
MODE: eval
|
||||
PROMPT_FILE:
|
||||
PROMPT_DATA: [ "a boy wearing a jacket", "a dog running on the lawn" ]
|
||||
IMAGE_SIZE: [ 1024, 1024 ]
|
||||
FIELDS: [ "prompt" ]
|
||||
DELIMITER: '#;#'
|
||||
PROMPT_PREFIX: ''
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
KEYS: [ 'index', 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 10
|
||||
SHOW_GPU_MEM: True
|
||||
-
|
||||
NAME: TensorboardLogHook
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 10000
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
SAVE_PROBE_PREFIX: 'image'
|
||||
@@ -0,0 +1,233 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
#
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 500
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 50
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/dit_sd3_1024
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
- NAME: "ModelscopeFs"
|
||||
TEMP_DIR: ./cache/cache_data
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionSD3
|
||||
PARAMETERIZATION: rf
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 1.5305
|
||||
SHIFT_FACTOR: 0.0609
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "shifted"
|
||||
"SHIFT": 3
|
||||
USE_EMA: False
|
||||
T_WEIGHT: uniform
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: MMDiT
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium@sd3_medium.safetensors
|
||||
IGNORE_KEYS: '^first_stage_model.'
|
||||
IN_CHANNELS: 16
|
||||
PATCH_SIZE: 2
|
||||
OUT_CHANNELS: 16
|
||||
DEPTH: 24
|
||||
INPUT_SIZE:
|
||||
ADM_IN_CHANNELS: 2048
|
||||
CONTEXT_EMBEDDER_CONFIG: { 'target': 'torch.nn.Linear', 'params': { 'in_features': 4096, 'out_features': 1536 } }
|
||||
NUM_PATCHES: 36864
|
||||
POS_EMBED_MAX_SIZE: 192
|
||||
POS_EMBED_SCALING_FACTOR:
|
||||
USE_CHECKPOINT: True
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium@sd3_medium.safetensors
|
||||
EMBED_DIM: 16
|
||||
IGNORE_KEYS: '^model.diffusion_model.'
|
||||
BATCH_SIZE: 1
|
||||
USE_CONV: False
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 16
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 16
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: SD3TextEmbedder
|
||||
P_ZERO: 0.0
|
||||
CLIP_L:
|
||||
NAME: FrozenCLIPEmbedder2
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@text_encoder
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@tokenizer
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: penultimate
|
||||
RETURN_POOLED: True
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
CLIP_G:
|
||||
NAME: FrozenCLIPEmbedder2
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@text_encoder_2
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@tokenizer_2
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: penultimate
|
||||
RETURN_POOLED: True
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
T5_XXL:
|
||||
NAME: T5EmbedderHF
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@text_encoder_3
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@tokenizer_3
|
||||
LENGTH: 256
|
||||
CLEAN: whitespace
|
||||
USE_GRAD: False
|
||||
T5_DTYPE: float16
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: euler
|
||||
SAMPLE_STEPS: 28
|
||||
SEED: 1749023094
|
||||
GUIDE_SCALE: 5.0
|
||||
GUIDE_RESCALE: 0.0
|
||||
DISCRETIZATION: trailing
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 1e-5
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 1024, 1024 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 1024, 1024 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: Text2ImageDataset
|
||||
MODE: eval
|
||||
PROMPT_FILE:
|
||||
PROMPT_DATA: [ "a cat holds a blackboard that writes \"hello world\"", "a dog running on the lawn" ]
|
||||
IMAGE_SIZE: [ 1024, 1024 ]
|
||||
FIELDS: [ "prompt" ]
|
||||
DELIMITER: '#;#'
|
||||
PROMPT_PREFIX: ''
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
KEYS: [ 'index', 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 10000
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 10
|
||||
SHOW_GPU_MEM: True
|
||||
-
|
||||
NAME: TensorboardLogHook
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 10000
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 50
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
SAVE_PROBE_PREFIX: 'image'
|
||||
@@ -0,0 +1,241 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
#
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 500
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 50
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/dit_sd3_1024_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
- NAME: "ModelscopeFs"
|
||||
TEMP_DIR: ./cache/cache_data
|
||||
#
|
||||
TUNER:
|
||||
- NAME: SwiftLoRA
|
||||
R: 128
|
||||
LORA_ALPHA: 128
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(.attn.qkv|.attn.proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionSD3
|
||||
PARAMETERIZATION: rf
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 1.5305
|
||||
SHIFT_FACTOR: 0.0609
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "shifted"
|
||||
"SHIFT": 3
|
||||
USE_EMA: False
|
||||
T_WEIGHT: uniform
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: MMDiT
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium@sd3_medium.safetensors
|
||||
IGNORE_KEYS: '^first_stage_model.'
|
||||
IN_CHANNELS: 16
|
||||
PATCH_SIZE: 2
|
||||
OUT_CHANNELS: 16
|
||||
DEPTH: 24
|
||||
INPUT_SIZE:
|
||||
ADM_IN_CHANNELS: 2048
|
||||
CONTEXT_EMBEDDER_CONFIG: { 'target': 'torch.nn.Linear', 'params': { 'in_features': 4096, 'out_features': 1536 } }
|
||||
NUM_PATCHES: 36864
|
||||
POS_EMBED_MAX_SIZE: 192
|
||||
POS_EMBED_SCALING_FACTOR:
|
||||
USE_CHECKPOINT: True
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium@sd3_medium.safetensors
|
||||
EMBED_DIM: 16
|
||||
IGNORE_KEYS: '^model.diffusion_model.'
|
||||
BATCH_SIZE: 1
|
||||
USE_CONV: False
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 16
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 16
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: SD3TextEmbedder
|
||||
P_ZERO: 0.0
|
||||
CLIP_L:
|
||||
NAME: FrozenCLIPEmbedder2
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@text_encoder
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@tokenizer
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: penultimate
|
||||
RETURN_POOLED: True
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
CLIP_G:
|
||||
NAME: FrozenCLIPEmbedder2
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@text_encoder_2
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@tokenizer_2
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: penultimate
|
||||
RETURN_POOLED: True
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
T5_XXL:
|
||||
NAME: T5EmbedderHF
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@text_encoder_3
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/stable-diffusion-3-medium-diffusers@tokenizer_3
|
||||
LENGTH: 256
|
||||
CLEAN: whitespace
|
||||
USE_GRAD: False
|
||||
T5_DTYPE: float16
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: euler
|
||||
SAMPLE_STEPS: 28
|
||||
SEED: 1749023094
|
||||
GUIDE_SCALE: 5.0
|
||||
GUIDE_RESCALE: 0.0
|
||||
DISCRETIZATION: trailing
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 5e-5
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bilinear
|
||||
SIZE: [ 1024, 1024 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCenterCrop
|
||||
SIZE: [ 1024, 1024 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: Text2ImageDataset
|
||||
MODE: eval
|
||||
PROMPT_FILE:
|
||||
PROMPT_DATA: [ "a cat holds a blackboard that writes \"hello world\"", "a dog running on the lawn" ]
|
||||
IMAGE_SIZE: [ 1024, 1024 ]
|
||||
FIELDS: [ "prompt" ]
|
||||
DELIMITER: '#;#'
|
||||
PROMPT_PREFIX: ''
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
KEYS: [ 'index', 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 10000
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 10
|
||||
SHOW_GPU_MEM: True
|
||||
-
|
||||
NAME: TensorboardLogHook
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 10000
|
||||
PRIORITY: 200
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
DISABLE_SNAPSHOT: True
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 50
|
||||
SAVE_LAST: True
|
||||
SAVE_NAME_PREFIX: 'step'
|
||||
SAVE_PROBE_PREFIX: 'image'
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd15_512_full
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
@@ -117,14 +118,14 @@ SOLVER:
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0064
|
||||
LEARNING_RATE: 0.00001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
@@ -190,7 +191,7 @@ SOLVER:
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd15_512_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
@@ -125,14 +126,14 @@ SOLVER:
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
@@ -198,7 +199,7 @@ SOLVER:
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
|
||||
@@ -0,0 +1,235 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd15_512_textlora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$"
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-v1-5@v1-5-pruned-emaonly.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS: 8
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 768
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: False
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: ClipTokenizer
|
||||
PRETRAINED_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
LENGTH: 77
|
||||
CLEAN: True
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenCLIPEmbedder
|
||||
FREEZE: True
|
||||
LAYER: last
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
USE_GRAD: True
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 512
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 512
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [512, 512]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
NAME: Select
|
||||
KEYS: ['prompt']
|
||||
META_KEYS: ['image_size']
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -0,0 +1,215 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd21_512_full
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-2-1-base@v2-1_512-ema-pruned.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 1024
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
PRETRAINED_MODEL:
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: [ ]
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: OpenClipTokenizer
|
||||
LENGTH: 77
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenOpenCLIPEmbedder
|
||||
ARCH: ViT-H-14
|
||||
PRETRAINED_MODEL:
|
||||
LAYER: penultimate
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.00001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 512
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 512
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [512, 512]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
NAME: Select
|
||||
KEYS: ['prompt']
|
||||
META_KEYS: ['image_size']
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -0,0 +1,224 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd21_512_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-2-1-base@v2-1_512-ema-pruned.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
SIZE_FACTOR: 8
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.012
|
||||
USE_EMA: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNet
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
MODEL_CHANNELS: 320
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
NUM_RES_BLOCKS: 2
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2, 1 ]
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
USE_CHECKPOINT: False
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 1
|
||||
CONTEXT_DIM: 1024
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
PRETRAINED_MODEL:
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: [ ]
|
||||
BATCH_SIZE: 4
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
TOKENIZER:
|
||||
NAME: OpenClipTokenizer
|
||||
LENGTH: 77
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: FrozenOpenCLIPEmbedder
|
||||
ARCH: ViT-H-14
|
||||
PRETRAINED_MODEL:
|
||||
LAYER: penultimate
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [512, 512]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.00001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 512
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 512
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [512, 512]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
NAME: Select
|
||||
KEYS: ['prompt']
|
||||
META_KEYS: ['image_size']
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd21_768_full
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusion
|
||||
@@ -113,14 +114,14 @@ SOLVER:
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [768, 768]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0064
|
||||
LEARNING_RATE: 0.00001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
@@ -186,7 +187,7 @@ SOLVER:
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd21_768_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
@@ -122,14 +123,14 @@ SOLVER:
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE:
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [768, 768]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
@@ -195,7 +196,7 @@ SOLVER:
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
-
|
||||
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_full
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionXL
|
||||
@@ -238,7 +239,7 @@ SOLVER:
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0064
|
||||
LEARNING_RATE: 0.00001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
@@ -306,7 +307,7 @@ SOLVER:
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_lora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
@@ -247,7 +248,7 @@ SOLVER:
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
@@ -315,7 +316,7 @@ SOLVER:
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
|
||||
@@ -0,0 +1,346 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 2000
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_textlora
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TUNER:
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "model.*(to_q|to_k|to_v|to_out.0|net.0.proj|net.2)$"
|
||||
-
|
||||
NAME: SwiftLoRA
|
||||
R: 64
|
||||
LORA_ALPHA: 64
|
||||
LORA_DROPOUT: 0.0
|
||||
BIAS: "none"
|
||||
TARGET_MODULES: "cond_stage_model.embedders.0.*(q_proj|k_proj|v_proj|out_proj|mlp.fc1|mlp.fc2)$"
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionXL
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-xl-base-1.0@sd_xl_base_1.0.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.13025
|
||||
SIZE_FACTOR: 8
|
||||
# DEFAULT_N_PROMPT: 'lowres, error, worst quality, low quality, jpeg artifacts, ugly, duplicate, morbid, mutilated, out of frame, extra fingers, mutated hands, poorly drawn hands, poorly drawn face, mutation, deformed, blurry, dehydrated, bad anatomy, bad proportions, extra limbs, cloned face, disfigured, gross proportions, malformed limbs, missing arms, missing legs, extra arms, extra legs, fused fingers, too many fingers, long neck, username, watermark, signature'
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.0120
|
||||
USE_EMA: False
|
||||
LOAD_REFINER: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 320
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: [ 1, 2, 10 ]
|
||||
CONTEXT_DIM: 2048
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2816
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 1
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
USE_GRAD: True
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenCLIPEmbedder
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: hidden
|
||||
LAYER_IDX: 11
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "target_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
REFINER_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 384
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 4
|
||||
CONTEXT_DIM: [ 1280, 1280, 1280, 1280 ]
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2560
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
#
|
||||
REFINER_COND_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "aesthetic_score" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 5.0
|
||||
GUIDE_RESCALE:
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [1024, 1024]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
INTERPOLATION: bicubic
|
||||
SIZE: 1024
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCropXL
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Select
|
||||
KEYS: [ 'img', 'prompt', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
META_KEYS: [ 'data_key', 'img_path' ]
|
||||
- NAME: Rename
|
||||
INPUT_KEY: [ 'img', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
OUTPUT_KEY: [ 'image', 'original_size_as_tuple', 'target_size_as_tuple', 'crop_coords_top_left' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_REMAP_KEYS: { 'Image': 'Target:FILE' }
|
||||
MS_DATASET_SPLIT: test_short
|
||||
OUTPUT_SIZE: [ 1024, 1024 ]
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 4
|
||||
NUM_WORKERS: 4
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
TRANSFORMS:
|
||||
- NAME: Select
|
||||
KEYS: [ 'prompt' ]
|
||||
META_KEYS: [ 'image_size' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
- NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
- NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
- NAME: CheckpointHook
|
||||
INTERVAL: 1000
|
||||
- NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
- NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd15_512_sce_ctr_hed
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
FREEZE:
|
||||
FREEZE_PART: [ "first_stage_model", "cond_stage_model", "model" ]
|
||||
@@ -127,7 +128,7 @@ SOLVER:
|
||||
DOWN_RATIO: 1.0
|
||||
CONTROL_ANNO:
|
||||
NAME: HedAnnotator
|
||||
PRETRAINED_MODEL: ms://damo/scepter_scedit@annotator/ckpts/ControlNetHED.pth
|
||||
PRETRAINED_MODEL: ms://iic/scepter_scedit@annotator/ckpts/ControlNetHED.pth
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
@@ -141,7 +142,7 @@ SOLVER:
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd21_768_sce_ctr_canny
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
FREEZE:
|
||||
FREEZE_PART: [ "first_stage_model", "cond_stage_model", "model" ]
|
||||
@@ -31,7 +32,7 @@ SOLVER:
|
||||
PARAMETERIZATION: v
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: True
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-2-1@v2-1_768-ema-pruned.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
@@ -140,7 +141,7 @@ SOLVER:
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sd21_768_sce_ctr_pose
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
FREEZE:
|
||||
FREEZE_PART: [ "first_stage_model", "cond_stage_model", "model" ]
|
||||
@@ -31,7 +32,7 @@ SOLVER:
|
||||
PARAMETERIZATION: v
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: True
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-2-1@v2-1_768-ema-pruned.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.18215
|
||||
@@ -125,8 +126,8 @@ SOLVER:
|
||||
DOWN_RATIO: 1.0
|
||||
CONTROL_ANNO:
|
||||
NAME: OpenposeAnnotator
|
||||
BODY_MODEL_PATH: ms://damo/scepter_scedit@annotator/ckpts/body_pose_model.pth
|
||||
HAND_MODEL_PATH: ms://damo/scepter_scedit@annotator/ckpts/hand_pose_model.pth
|
||||
BODY_MODEL_PATH: ms://iic/scepter_scedit@annotator/ckpts/body_pose_model.pth
|
||||
HAND_MODEL_PATH: ms://iic/scepter_scedit@annotator/ckpts/hand_pose_model.pth
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
@@ -140,7 +141,7 @@ SOLVER:
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
|
||||
@@ -0,0 +1,379 @@
|
||||
ENV:
|
||||
BACKEND: nccl
|
||||
SOLVER:
|
||||
NAME: LatentDiffusionSolver
|
||||
RESUME_FROM:
|
||||
LOAD_MODEL_ONLY: True
|
||||
USE_FSDP: False
|
||||
SHARDING_STRATEGY:
|
||||
USE_AMP: True
|
||||
DTYPE: float16
|
||||
CHANNELS_LAST: True
|
||||
MAX_STEPS: 200
|
||||
MAX_EPOCHS: -1
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_sce_ctr_canny
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
FREEZE:
|
||||
FREEZE_PART: [ "first_stage_model", "cond_stage_model", "model" ]
|
||||
TRAIN_PART: [ "control_blocks" ]
|
||||
#
|
||||
MODEL:
|
||||
NAME: LatentDiffusionXLSCEControl
|
||||
PARAMETERIZATION: eps
|
||||
TIMESTEPS: 1000
|
||||
MIN_SNR_GAMMA:
|
||||
ZERO_TERMINAL_SNR: False
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/stable-diffusion-xl-base-1.0@sd_xl_base_1.0.safetensors
|
||||
IGNORE_KEYS: [ ]
|
||||
SCALE_FACTOR: 0.13025
|
||||
SIZE_FACTOR: 8
|
||||
DEFAULT_N_PROMPT:
|
||||
SCHEDULE_ARGS:
|
||||
"NAME": "scaled_linear"
|
||||
"BETA_MIN": 0.00085
|
||||
"BETA_MAX": 0.0120
|
||||
USE_EMA: False
|
||||
LOAD_REFINER: False
|
||||
#
|
||||
DIFFUSION_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 320
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: [ 1, 2, 10 ]
|
||||
CONTEXT_DIM: 2048
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2816
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
#
|
||||
FIRST_STAGE_MODEL:
|
||||
NAME: AutoencoderKL
|
||||
EMBED_DIM: 4
|
||||
PRETRAINED_MODEL:
|
||||
IGNORE_KEYS: []
|
||||
BATCH_SIZE: 1
|
||||
#
|
||||
ENCODER:
|
||||
NAME: Encoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DOUBLE_Z: True
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
#
|
||||
DECODER:
|
||||
NAME: Decoder
|
||||
CH: 128
|
||||
OUT_CH: 3
|
||||
NUM_RES_BLOCKS: 2
|
||||
IN_CHANNELS: 3
|
||||
ATTN_RESOLUTIONS: [ ]
|
||||
CH_MULT: [ 1, 2, 4, 4 ]
|
||||
Z_CHANNELS: 4
|
||||
DROPOUT: 0.0
|
||||
RESAMP_WITH_CONV: True
|
||||
GIVE_PRE_END: False
|
||||
TANH_OUT: False
|
||||
#
|
||||
COND_STAGE_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenCLIPEmbedder
|
||||
PRETRAINED_MODEL: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
TOKENIZER_PATH: ms://AI-ModelScope/clip-vit-large-patch14
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
LAYER: hidden
|
||||
LAYER_IDX: 11
|
||||
USE_FINAL_LAYER_NORM: False
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "target_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
REFINER_MODEL:
|
||||
NAME: DiffusionUNetXL
|
||||
PRETRAINED_MODEL:
|
||||
IN_CHANNELS: 4
|
||||
OUT_CHANNELS: 4
|
||||
NUM_RES_BLOCKS: 2
|
||||
MODEL_CHANNELS: 384
|
||||
ATTENTION_RESOLUTIONS: [ 4, 2 ]
|
||||
DROPOUT: 0
|
||||
CHANNEL_MULT: [ 1, 2, 4, 4 ]
|
||||
CONV_RESAMPLE: True
|
||||
DIMS: 2
|
||||
NUM_CLASSES: sequential
|
||||
USE_CHECKPOINT: False
|
||||
NUM_HEADS: -1
|
||||
NUM_HEADS_CHANNELS: 64
|
||||
USE_SCALE_SHIFT_NORM: False
|
||||
RESBLOCK_UPDOWN: False
|
||||
USE_NEW_ATTENTION_ORDER: True
|
||||
USE_SPATIAL_TRANSFORMER: True
|
||||
TRANSFORMER_DEPTH: 4
|
||||
CONTEXT_DIM: [ 1280, 1280, 1280, 1280 ]
|
||||
DISABLE_MIDDLE_SELF_ATTN: False
|
||||
USE_LINEAR_IN_TRANSFORMER: True
|
||||
ADM_IN_CHANNELS: 2560
|
||||
USE_SENTENCE_EMB: False
|
||||
USE_WORD_MAPPING: False
|
||||
REFINER_COND_MODEL:
|
||||
NAME: GeneralConditioner
|
||||
PRETRAINED_MODEL:
|
||||
EMBEDDERS:
|
||||
-
|
||||
NAME: FrozenOpenCLIPEmbedder2
|
||||
ARCH: ViT-bigG-14
|
||||
PRETRAINED_MODEL:
|
||||
MAX_LENGTH: 77
|
||||
FREEZE: True
|
||||
ALWAYS_RETURN_POOLED: True
|
||||
LEGACY: False
|
||||
LAYER: penultimate
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "prompt" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "original_size_as_tuple" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "crop_coords_top_left" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
-
|
||||
NAME: ConcatTimestepEmbedderND
|
||||
OUT_DIM: 256
|
||||
IS_TRAINABLE: False
|
||||
UCG_RATE: 0.0
|
||||
INPUT_KEYS: [ "aesthetic_score" ]
|
||||
LEGACY_UCG_VALUE:
|
||||
#
|
||||
LOSS:
|
||||
NAME: ReconstructLoss
|
||||
LOSS_TYPE: l2
|
||||
#
|
||||
CONTROL_MODEL:
|
||||
NAME: CSCTuners
|
||||
PRE_HINT_IN_CHANNELS: 3
|
||||
PRE_HINT_OUT_CHANNELS: 320
|
||||
DENSE_HINT_KERNAL: 3
|
||||
PRE_HINT_DIM_RATIO: 2.0
|
||||
SCALE: 1.0
|
||||
SC_TUNER_CFG:
|
||||
NAME: SCTuner
|
||||
TUNER_NAME: SCEAdapter
|
||||
DOWN_RATIO: 1.0
|
||||
CONTROL_ANNO:
|
||||
NAME: CannyAnnotator
|
||||
#
|
||||
SAMPLE_ARGS:
|
||||
SAMPLER: ddim
|
||||
SAMPLE_STEPS: 50
|
||||
SEED: 2023
|
||||
GUIDE_SCALE: 7.5
|
||||
GUIDE_RESCALE: 0.5
|
||||
DISCRETIZATION: trailing
|
||||
IMAGE_SIZE: [1024, 1024]
|
||||
RUN_TRAIN_N: False
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
AMSGRAD: False
|
||||
#
|
||||
TRAIN_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: train
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 1
|
||||
NUM_WORKERS: 4
|
||||
SAMPLER:
|
||||
NAME: LoopSampler
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleResize
|
||||
SIZE: 1024
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: FlexibleCropXL
|
||||
SIZE: 1024
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ToNumpy
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image_preprocess' ]
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Rename
|
||||
INPUT_KEY: [ 'img', 'image_preprocess', 'img_original_size_as_tuple', 'img_target_size_as_tuple', 'img_crop_coords_top_left' ]
|
||||
OUTPUT_KEY: [ 'image', 'image_preprocess', 'original_size_as_tuple', 'target_size_as_tuple', 'crop_coords_top_left' ]
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt', 'image_preprocess', 'original_size_as_tuple', 'target_size_as_tuple', 'crop_coords_top_left' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
EVAL_DATA:
|
||||
NAME: ImageTextPairMSDataset
|
||||
MODE: eval
|
||||
MS_DATASET_NAME: style_custom_dataset
|
||||
MS_DATASET_NAMESPACE: damo
|
||||
MS_DATASET_SUBNAME: 3D
|
||||
PROMPT_PREFIX: ""
|
||||
MS_DATASET_SPLIT: train_short
|
||||
MS_REMAP_KEYS: { 'Image:FILE': 'Target:FILE' }
|
||||
REPLACE_STYLE: False
|
||||
PIN_MEMORY: True
|
||||
BATCH_SIZE: 10
|
||||
NUM_WORKERS: 4
|
||||
TRANSFORMS:
|
||||
- NAME: LoadImageFromFile
|
||||
RGB_ORDER: RGB
|
||||
BACKEND: pillow
|
||||
- NAME: Resize
|
||||
SIZE: 1024
|
||||
INTERPOLATION: bilinear
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: CenterCrop
|
||||
SIZE: 1024
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: ToNumpy
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'image_preprocess' ]
|
||||
- NAME: ImageToTensor
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: pillow
|
||||
- NAME: Normalize
|
||||
MEAN: [ 0.5, 0.5, 0.5 ]
|
||||
STD: [ 0.5, 0.5, 0.5 ]
|
||||
INPUT_KEY: [ 'img' ]
|
||||
OUTPUT_KEY: [ 'img' ]
|
||||
BACKEND: torchvision
|
||||
- NAME: Rename
|
||||
INPUT_KEY: [ 'img', 'image_preprocess' ]
|
||||
OUTPUT_KEY: [ 'image', 'image_preprocess' ]
|
||||
- NAME: Select
|
||||
KEYS: [ 'image', 'prompt', 'image_preprocess' ]
|
||||
META_KEYS: [ 'data_key' ]
|
||||
#
|
||||
TRAIN_HOOKS:
|
||||
-
|
||||
NAME: BackwardHook
|
||||
PRIORITY: 0
|
||||
-
|
||||
NAME: LogHook
|
||||
LOG_INTERVAL: 50
|
||||
-
|
||||
NAME: CheckpointHook
|
||||
INTERVAL: 100
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
#
|
||||
EVAL_HOOKS:
|
||||
-
|
||||
NAME: ProbeDataHook
|
||||
PROB_INTERVAL: 100
|
||||
@@ -14,13 +14,14 @@ SOLVER:
|
||||
NUM_FOLDS: 1
|
||||
ACCU_STEP: 1
|
||||
EVAL_INTERVAL: 100
|
||||
RESCALE_LR: False
|
||||
#
|
||||
WORK_DIR: ./cache/save_data/sdxl_1024_sce_ctr_color
|
||||
LOG_FILE: std_log.txt
|
||||
#
|
||||
FILE_SYSTEM:
|
||||
NAME: "ModelscopeFs"
|
||||
TEMP_DIR: "./cache/data"
|
||||
TEMP_DIR: "./cache/cache_data"
|
||||
#
|
||||
FREEZE:
|
||||
FREEZE_PART: [ "first_stage_model", "cond_stage_model", "model" ]
|
||||
@@ -231,8 +232,9 @@ SOLVER:
|
||||
CONTROL_MODEL:
|
||||
NAME: CSCTuners
|
||||
PRE_HINT_IN_CHANNELS: 3
|
||||
PRE_HINT_OUT_CHANNELS: 256
|
||||
PRE_HINT_OUT_CHANNELS: 320
|
||||
DENSE_HINT_KERNAL: 3
|
||||
PRE_HINT_DIM_RATIO: 2.0
|
||||
SCALE: 1.0
|
||||
SC_TUNER_CFG:
|
||||
NAME: SCTuner
|
||||
@@ -254,7 +256,7 @@ SOLVER:
|
||||
#
|
||||
OPTIMIZER:
|
||||
NAME: AdamW
|
||||
LEARNING_RATE: 0.064
|
||||
LEARNING_RATE: 0.0001
|
||||
BETAS: [ 0.9, 0.999 ]
|
||||
EPS: 1e-8
|
||||
WEIGHT_DECAY: 1e-2
|
||||
|
||||