diff --git a/README.md b/README.md
index 74f7e9e..95c6a4c 100644
--- a/README.md
+++ b/README.md
@@ -35,11 +35,119 @@
+## 📚 Introduction
+
+The original intention behind the design of ACE++ was to unify reference image generation, local editing,
+and controllable generation into a single framework, and to enable one model to adapt to a wider range of tasks.
+A more versatile model is often capable of handling more complex tasks. We have already released three LoRA models,
+focusing on portraits, objects, and regional editing, with the expectation that each would demonstrate strong adaptability
+within their respective domains. Undoubtedly, this presents certain challenges.
+
+We are currently training a fully fine-tuned model, which has now entered the final stage of quality tuning.
+We are confident it will be released soon. This model will support a broader range of capabilities and is
+expected to empower community developers to build even more interesting applications.
+
## 📢 News
- [x] **[2025.01.06]** Release the code and models of ACE++.
- [x] **[2025.01.07]** Release the demo on [HuggingFace](https://huggingface.co/spaces/scepter-studio/ACE-Plus).
- [x] **[2025.01.16]** Release the training code for lora.
-- [] **[ToDo]** Update Models.
+- [x] **[2025.02.15]** Collection of workflows in Comfyui.
+- [x] **[2025.02.15]** Release the config for fully fine-tuning.
+- [] **[ToDo]** Release a unified fft model for ACE++, support more image to image tasks.
+
+## 🔥 Comfyui Workflows in community
+We are deeply grateful to the community developers for building many fascinating applications based on the ACE++ series of models.
+During this process, we have received valuable feedback, particularly regarding artifacts in generated images and the stability of the results.
+In response to these issues, many developers have proposed creative solutions, which have greatly inspired us, and we pay tribute to them.
+At the same time, we will take these concerns into account in our further optimization efforts, carefully evaluating and testing before releasing new models.
+
+In the table below, we have briefly listed some workflows for everyone to use.
+
+
+
+Additionally, many bloggers have published tutorials on how to use it, which are listed in the table below.
+
+
+
## 🔥 ACE Models
ACE++ provides a comprehensive toolkit for image editing and generation to support various applications. We encourage developers to choose the appropriate model based on their own scenarios and to fine-tune their models using data from their specific scenarios to achieve more stable results.
diff --git a/assets/comfyui/chumen_tryon.jpg b/assets/comfyui/chumen_tryon.jpg
new file mode 100644
index 0000000..3a57fa3
Binary files /dev/null and b/assets/comfyui/chumen_tryon.jpg differ
diff --git a/assets/comfyui/feixiangjing_face.png b/assets/comfyui/feixiangjing_face.png
new file mode 100644
index 0000000..1bf2fa9
Binary files /dev/null and b/assets/comfyui/feixiangjing_face.png differ
diff --git a/assets/comfyui/haobeen_ace_plus.jpg b/assets/comfyui/haobeen_ace_plus.jpg
new file mode 100644
index 0000000..d458345
Binary files /dev/null and b/assets/comfyui/haobeen_ace_plus.jpg differ
diff --git a/assets/comfyui/jax_face_swap.jpg b/assets/comfyui/jax_face_swap.jpg
new file mode 100644
index 0000000..9f1f6af
Binary files /dev/null and b/assets/comfyui/jax_face_swap.jpg differ
diff --git a/assets/comfyui/leeguandong_subject.jpg b/assets/comfyui/leeguandong_subject.jpg
new file mode 100644
index 0000000..9a5a85f
Binary files /dev/null and b/assets/comfyui/leeguandong_subject.jpg differ
diff --git a/assets/comfyui/t8_star_face.jpg b/assets/comfyui/t8_star_face.jpg
new file mode 100644
index 0000000..0abf26f
Binary files /dev/null and b/assets/comfyui/t8_star_face.jpg differ
diff --git a/assets/comfyui/t8_star_logo.jpg b/assets/comfyui/t8_star_logo.jpg
new file mode 100644
index 0000000..178a5bc
Binary files /dev/null and b/assets/comfyui/t8_star_logo.jpg differ
diff --git a/assets/comfyui/t8_star_tryon.jpg b/assets/comfyui/t8_star_tryon.jpg
new file mode 100644
index 0000000..fc5d0f4
Binary files /dev/null and b/assets/comfyui/t8_star_tryon.jpg differ
diff --git a/inference/ace_plus_diffusers.py b/inference/ace_plus_diffusers.py
index 2ae9f35..7b1e9fa 100644
--- a/inference/ace_plus_diffusers.py
+++ b/inference/ace_plus_diffusers.py
@@ -12,7 +12,6 @@ from scepter.modules.utils.logger import get_logger
from transformers import T5TokenizerFast
from .utils import ACEPlusImageProcessor
-
class ACEPlusDiffuserInference():
def __init__(self, logger=None):
if logger is None:
@@ -39,7 +38,6 @@ class ACEPlusDiffuserInference():
self.pipe.tokenizer_2 = tokenizer_2
self.load_default(cfg.DEFAULT_PARAS)
-
def prepare_input(self,
image,
mask,
@@ -101,6 +99,8 @@ class ACEPlusDiffuserInference():
with FS.get_from(lora_path) as local_path:
self.pipe.load_lora_weights(local_path)
+
+
image = self.pipe(
prompt=prompt,
masked_image_latents=masked_image_latents,
diff --git a/train_config/ace_plus_fft.yaml b/train_config/ace_plus_fft.yaml
new file mode 100644
index 0000000..c63f5fe
--- /dev/null
+++ b/train_config/ace_plus_fft.yaml
@@ -0,0 +1,268 @@
+ENV:
+ BACKEND: nccl
+ SEED: 1999
+SOLVER:
+ # NAME DESCRIPTION: TYPE: default: 'LatentUfitSolver'
+ NAME: ACEPlusSolver
+ # MAX_STEPS DESCRIPTION: The total steps for training. TYPE: int default: 100000
+ MAX_STEPS: 100000
+ # USE_AMP DESCRIPTION: Use amp to surpport mix precision or not, default is False. TYPE: bool default: False
+ USE_AMP: True
+ # DTYPE DESCRIPTION: The precision for training. TYPE: str default: 'float32'
+ DTYPE: bfloat16
+ ENABLE_GRADSCALER: False
+ # USE_FAIRSCALE DESCRIPTION: Use fairscale as the backend of ddp, default False. TYPE: bool default: False
+ USE_FAIRSCALE: False
+ USE_ORIG_PARAMS: True
+ USE_FSDP: True # lora use ddp(USE_FSDP=False), else use fsdp(USE_FSDP=True)
+ # LOAD_MODEL_ONLY DESCRIPTION: Only load the model rather than the optimizer and schedule, default is False. TYPE: bool default: False
+ LOAD_MODEL_ONLY: False
+ # RESUME_FROM DESCRIPTION: Resume from some state of training! TYPE: str default: ''
+ RESUME_FROM:
+ # WORK_DIR DESCRIPTION: Save dir of the training log or model. TYPE: str default: ''
+ WORK_DIR: ./examples/exp_example/
+ # LOG_FILE DESCRIPTION: Save log path. TYPE: str default: ''
+ LOG_FILE: std_log.txt
+ # LOG_TRAIN_NUM DESCRIPTION: The number samples used to log in training phase. TYPE: int default: -1
+ LOG_TRAIN_NUM: 16
+ # FSDP_REDUCE_DTYPE DESCRIPTION: The dtype of reduce in FSDP. TYPE: str default: 'float16'
+ FSDP_REDUCE_DTYPE: float32
+ # FSDP_BUFFER_DTYPE DESCRIPTION: The dtype of buffer in FSDP. TYPE: str default: 'float16'
+ FSDP_BUFFER_DTYPE: float32
+ # FSDP_SHARD_MODULES DESCRIPTION: The modules to be sharded in FSDP. TYPE: list default: ['model']
+ FSDP_SHARD_MODULES:
+ - MODULE: 'model'
+ FSDP_GROUP: [ 'single_blocks', 'double_blocks']
+ - MODULE: 'cond_stage_model.t5_model.hf_module.encoder'
+ FSDP_GROUP: [ 'block' ] #
+ SAVE_MODULES: [ 'model'] #
+ TRAIN_MODULES: ['model']
+
+ #
+ FILE_SYSTEM:
+ - NAME: HuggingfaceFs
+ TEMP_DIR: ./cache
+ - NAME: ModelscopeFs
+ TEMP_DIR: ./cache
+ #
+ MODEL:
+ NAME: LatentDiffusionACEPlus
+ PARAMETERIZATION: rf
+ TIMESTEPS: 1000
+ GUIDE_SCALE: 1.0
+ PRETRAINED_MODEL:
+ IGNORE_KEYS: [ ]
+ USE_EMA: False
+ EVAL_EMA: False
+ SIZE_FACTOR: 8
+ DIFFUSION:
+ NAME: DiffusionFluxRF
+ PREDICTION_TYPE: raw
+ NOISE_NORM: True
+ # NOISE_SCHEDULER DESCRIPTION: TYPE: default: ''
+ NOISE_SCHEDULER:
+ NAME: FlowMatchFluxShiftScheduler
+ SHIFT: False
+ PRE_T_SAMPLE: True
+ PRE_T_SAMPLE_FOLD: 1
+ SIGMOID_SCALE: 1
+ BASE_SHIFT: 0.5
+ MAX_SHIFT: 1.15
+ SAMPLER_SCHEDULER:
+ NAME: FlowMatchFluxShiftScheduler
+ SHIFT: True
+ PRE_T_SAMPLE: False
+ SIGMOID_SCALE: 1
+ BASE_SHIFT: 0.5
+ MAX_SHIFT: 1.15
+
+ #
+ DIFFUSION_MODEL:
+ # NAME DESCRIPTION: TYPE: default: 'Flux'
+ NAME: FluxMRACEPlus
+ PRETRAINED_MODEL: ${FLUX_FILL_PATH}/flux1-fill-dev.safetensors
+ # IN_CHANNELS DESCRIPTION: model's input channels. TYPE: int default: 64
+ IN_CHANNELS: 384
+ # OUT_CHANNELS DESCRIPTION: model's input channels. TYPE: int default: 64
+ OUT_CHANNELS: 64
+ # HIDDEN_SIZE DESCRIPTION: model's hidden size. TYPE: int default: 1024
+ HIDDEN_SIZE: 3072
+ REDUX_DIM: 1152
+ # NUM_HEADS DESCRIPTION: number of heads in the transformer. TYPE: int default: 16
+ NUM_HEADS: 24
+ # AXES_DIM DESCRIPTION: dimensions of the axes of the positional encoding. TYPE: list default: [16, 56, 56]
+ AXES_DIM: [ 16, 56, 56 ]
+ # THETA DESCRIPTION: theta for positional encoding. TYPE: int default: 10000
+ THETA: 10000
+ # VEC_IN_DIM DESCRIPTION: dimension of the vector input. TYPE: int default: 768
+ VEC_IN_DIM: 768
+ # GUIDANCE_EMBED DESCRIPTION: whether to use guidance embedding. TYPE: bool default: False
+ GUIDANCE_EMBED: True
+ # CONTEXT_IN_DIM DESCRIPTION: dimension of the context input. TYPE: int default: 4096
+ CONTEXT_IN_DIM: 4096
+ # MLP_RATIO DESCRIPTION: ratio of mlp hidden size to hidden size. TYPE: float default: 4.0
+ MLP_RATIO: 4.0
+ # QKV_BIAS DESCRIPTION: whether to use bias in qkv projection. TYPE: bool default: True
+ QKV_BIAS: True
+ # DEPTH DESCRIPTION: number of transformer blocks. TYPE: int default: 19
+ DEPTH: 19
+ # DEPTH_SINGLE_BLOCKS DESCRIPTION: number of transformer blocks in the single stream block. TYPE: int default: 38
+ DEPTH_SINGLE_BLOCKS: 38
+ ATTN_BACKEND: flash_attn
+
+ #
+ FIRST_STAGE_MODEL:
+ NAME: AutoencoderKLFlux
+ EMBED_DIM: 16
+ PRETRAINED_MODEL: ${FLUX_FILL_PATH}/ae.safetensors
+ IGNORE_KEYS: [ ]
+ BATCH_SIZE: 8
+ USE_CONV: False
+ SCALE_FACTOR: 0.3611
+ SHIFT_FACTOR: 0.1159
+ #
+ ENCODER:
+ NAME: Encoder
+ CH: 128
+ OUT_CH: 3
+ NUM_RES_BLOCKS: 2
+ IN_CHANNELS: 3
+ ATTN_RESOLUTIONS: [ ]
+ CH_MULT: [ 1, 2, 4, 4 ]
+ Z_CHANNELS: 16
+ DOUBLE_Z: True
+ DROPOUT: 0.0
+ RESAMP_WITH_CONV: True
+ #
+ DECODER:
+ NAME: Decoder
+ CH: 128
+ OUT_CH: 3
+ NUM_RES_BLOCKS: 2
+ IN_CHANNELS: 3
+ ATTN_RESOLUTIONS: [ ]
+ CH_MULT: [ 1, 2, 4, 4 ]
+ Z_CHANNELS: 16
+ DROPOUT: 0.0
+ RESAMP_WITH_CONV: True
+ GIVE_PRE_END: False
+ TANH_OUT: False
+ #
+ COND_STAGE_MODEL:
+ # NAME DESCRIPTION: TYPE: default: 'T5PlusClipFluxEmbedder'
+ NAME: T5ACEPlusClipFluxEmbedder
+ # T5_MODEL DESCRIPTION: TYPE: default: ''
+ T5_MODEL:
+ # NAME DESCRIPTION: TYPE: default: 'HFEmbedder'
+ NAME: ACEHFEmbedder
+ # HF_MODEL_CLS DESCRIPTION: huggingface cls in transfomer TYPE: NoneType default: None
+ HF_MODEL_CLS: T5EncoderModel
+ # MODEL_PATH DESCRIPTION: model folder path TYPE: NoneType default: None
+ MODEL_PATH: ${FLUX_FILL_PATH}/text_encoder_2/
+ # HF_TOKENIZER_CLS DESCRIPTION: huggingface cls in transfomer TYPE: NoneType default: None
+ HF_TOKENIZER_CLS: T5Tokenizer
+ # TOKENIZER_PATH DESCRIPTION: tokenizer folder path TYPE: NoneType default: None
+ TOKENIZER_PATH: ${FLUX_FILL_PATH}/tokenizer_2/
+ ADDED_IDENTIFIER: [ '
','{image}', '{caption}', '{mask}', '{ref_image}', '{image1}', '{image2}', '{image3}', '{image4}', '{image5}', '{image6}', '{image7}', '{image8}', '{image9}' ]
+ # MAX_LENGTH DESCRIPTION: max length of input TYPE: int default: 77
+ MAX_LENGTH: 512
+ # OUTPUT_KEY DESCRIPTION: output key TYPE: str default: 'last_hidden_state'
+ OUTPUT_KEY: last_hidden_state
+ # D_TYPE DESCRIPTION: dtype TYPE: str default: 'bfloat16'
+ D_TYPE: bfloat16
+ # BATCH_INFER DESCRIPTION: batch infer TYPE: bool default: False
+ BATCH_INFER: False
+ CLEAN: whitespace
+ # CLIP_MODEL DESCRIPTION: TYPE: default: ''
+ CLIP_MODEL:
+ # NAME DESCRIPTION: TYPE: default: 'HFEmbedder'
+ NAME: ACEHFEmbedder
+ # HF_MODEL_CLS DESCRIPTION: huggingface cls in transfomer TYPE: NoneType default: None
+ HF_MODEL_CLS: CLIPTextModel
+ # MODEL_PATH DESCRIPTION: model folder path TYPE: NoneType default: None
+ MODEL_PATH: ${FLUX_FILL_PATH}/text_encoder/
+ # HF_TOKENIZER_CLS DESCRIPTION: huggingface cls in transfomer TYPE: NoneType default: None
+ HF_TOKENIZER_CLS: CLIPTokenizer
+ # TOKENIZER_PATH DESCRIPTION: tokenizer folder path TYPE: NoneType default: None
+ TOKENIZER_PATH: ${FLUX_FILL_PATH}/tokenizer/
+ # MAX_LENGTH DESCRIPTION: max length of input TYPE: int default: 77
+ MAX_LENGTH: 77
+ # OUTPUT_KEY DESCRIPTION: output key TYPE: str default: 'last_hidden_state'
+ OUTPUT_KEY: pooler_output
+ # D_TYPE DESCRIPTION: dtype TYPE: str default: 'bfloat16'
+ D_TYPE: bfloat16
+ # BATCH_INFER DESCRIPTION: batch infer TYPE: bool default: False
+ BATCH_INFER: True
+ CLEAN: whitespace
+ #
+ SAMPLE_ARGS:
+ SAMPLE_STEPS: 28
+ SAMPLER: flow_euler
+ SEED: 42
+ IMAGE_SIZE: [ 1024, 1024 ]
+ GUIDE_SCALE: 50
+
+ LR_SCHEDULER:
+ NAME: StepAnnealingLR
+ WARMUP_STEPS: 0
+ TOTAL_STEPS: 100000
+ DECAY_MODE: 'cosine'
+ #
+ OPTIMIZER:
+ NAME: AdamW
+ LEARNING_RATE: 5e-5
+ BETAS: [ 0.9, 0.999 ]
+ EPS: 1e-6
+ WEIGHT_DECAY: 1e-2
+ AMSGRAD: False
+ #
+ TRAIN_DATA:
+ NAME: ACEPlusDataset
+ MODE: train
+ DATA_LIST: data/train.csv
+ DELIMITER: "#;#"
+ # input_image, input_mask, input_reference_image, target_image, instruction, task_type
+ FIELDS: ["edit_image", "edit_mask", "ref_image", "target_image", "prompt", "data_type"]
+ PATH_PREFIX: ""
+ EDIT_TYPE_LIST: []
+ MAX_SEQ_LEN: 2048
+ D: 16
+ PIN_MEMORY: True
+ BATCH_SIZE: 1
+ NUM_WORKERS: 4
+ SAMPLER:
+ NAME: LoopSampler
+
+ EVAL_DATA:
+ NAME: ACEPlusDataset
+ MODE: eval
+ DATA_LIST: data/train.csv
+ DELIMITER: "#;#"
+ # input_image, input_mask, input_reference_image, target_image, instruction, task_type
+ FIELDS: [ "edit_image", "edit_mask", "ref_image", "target_image", "prompt", "data_type" ]
+ PATH_PREFIX: ""
+ EDIT_TYPE_LIST: [ ]
+ MAX_SEQ_LEN: 2048
+ D: 16
+ PIN_MEMORY: True
+ BATCH_SIZE: 1
+ NUM_WORKERS: 4
+
+ TRAIN_HOOKS:
+ - NAME: ACEBackwardHook
+ GRADIENT_CLIP: 1.0
+ PRIORITY: 10
+ - NAME: LogHook
+ LOG_INTERVAL: 20
+ - NAME: ACECheckpointHook
+ INTERVAL: 250
+ PRIORITY: 200
+ - NAME: ProbeDataHook
+ PROB_INTERVAL: 50
+ PRIORITY: 0
+ - NAME: TensorboardLogHook
+ LOG_INTERVAL: 50
+ EVAL_HOOKS:
+ - NAME: ProbeDataHook
+ PROB_INTERVAL: 50
+ PRIORITY: 0
\ No newline at end of file