add auto-gpu picking

This commit is contained in:
aiXander
2024-03-30 20:37:01 -07:00
parent d56af4c5ed
commit 90feb0709c
6 changed files with 35 additions and 25 deletions
+2 -3
View File
@@ -22,7 +22,7 @@ Code / Cleanup:
- Modularize the logic in train.py as much as possible, trying to minimize dev work that needs to happen when SD3 drops (in progress)
- ~~make a clean train.py entrypoint that can be run as a normal python command (instead of having to use cog)~~
- ~~make it so the textual_inversion optimizer only optimizes the actual trained token embeddings instead of all of them + resetting later~~
- test if the trained concepts with peft are compatible with ComfyUI / AUTO1111
- Make sure the trained concepts (with peft) are compatible with ComfyUI / AUTO1111
Algo:
- Add aspect_ratio bucketing into the dataloader so we can train on non-square images (take this from https://github.com/kohya-ss/sd-scripts)
@@ -30,7 +30,6 @@ Algo:
- the random initialization of the token embeddings has a relatively large impact on the final outcome, there are prob ways to reduce
this random variance, eg CLIP_similarity pretraining.
- Improve the img captioning by swapping BLIP for cogVLM: https://github.com/THUDM/CogVLM
- add gradient clipping, see https://github.com/cloneofsimo/lora/blob/master/lora_diffusion/cli_lora_pti.py#L606C13-L608C14
Bugfixing:
see msgs at: https://discord.com/channels/573691888050241543/1184175211998883950/1217550596878373037
@@ -38,10 +37,10 @@ see msgs at: https://discord.com/channels/573691888050241543/1184175211998883950
- Try to find out why the diffusers training script works for sd15 and ours doesnt:
See here: https://huggingface.co/blog/sdxl_lora_advanced_script
and here: https://github.com/huggingface/diffusers/tree/main/examples/advanced_diffusion_training
- figure out how to adaptively set lora_scale at inference time using peft + diffusers? (https://github.com/huggingface/peft/blob/main/src/peft/tuners/lora/layer.py#L240)
Bigger improvements:
- add stronger token regularization (eg CelebBasis spanning basis)
- Add multi-token training
- pre-optimize token embeddings using CLIP-similarity (cfr aesthetic gradients: https://github.com/vicgalle/stable-diffusion-aesthetic-gradients/tree/main)
- implement perfusion: https://research.nvidia.com/labs/par/Perfusion/
+6 -10
View File
@@ -3,7 +3,7 @@ from trainer.utils.lora import patch_pipe_with_lora, blend_conditions
from trainer.utils.val_prompts import val_prompts
from trainer.utils.prompt import prepare_prompt_for_lora
from trainer.utils.io import make_validation_img_grid
from trainer.dataset_and_utils import seed_everything
from trainer.dataset_and_utils import seed_everything, pick_best_gpu_id
from diffusers import EulerDiscreteScheduler
import torch
@@ -11,23 +11,18 @@ from huggingface_hub import hf_hub_download
import os, json, random, time
if __name__ == "__main__":
pretrained_model = pretrained_models['sdxl']
lora_path = 'lora_models/clipx_tiny_test---sdxl_style_lora/checkpoints/checkpoint-500'
lora_scale = 0.75
lora_path = 'lora_models/gene---sdxl_face_lora/checkpoints/checkpoint-600'
lora_scale = 0.7
modulate_token_strength = True
seed = 0
render_size = (1024, 1024) # W,H
n_imgs = 4
n_steps = 30
guidance_scale = 8
use_lightning = False
@@ -37,6 +32,7 @@ if __name__ == "__main__":
os.makedirs(output_dir, exist_ok=True)
seed_everything(seed)
pick_best_gpu_id()
(pipe,
tokenizer_one,
+3 -3
View File
@@ -31,7 +31,6 @@ class TrainingConfig(BaseModel):
caption_model: Literal["gpt4-v", "blip"] = "blip"
left_right_flip_augmentation: bool = True
augment_imgs_up_to_n: int = 20
n_tokens: int = 2
mask_target_prompts: Union[None, str] = None
crop_based_on_salience: bool = True
use_face_detection_instead: bool = False
@@ -43,8 +42,9 @@ class TrainingConfig(BaseModel):
off_ratio_power: float = False
allow_tf32: bool = True
mixed_precision: Literal["fp16", "bf16", "fp32"] = "bf16"
inserting_list_tokens: List[str] = ["<s0>"]
token_dict: dict = {"TOKEN": "<s0>"}
n_tokens: int = 2
inserting_list_tokens: List[str] = ["<s0>","<s1>"]
token_dict: dict = {"TOKEN": "<s0><s1>"}
device: str = "cuda:0"
crops_coords_top_left_h: int = 0
crops_coords_top_left_w: int = 0
+21
View File
@@ -17,6 +17,27 @@ from transformers import AutoTokenizer, PretrainedConfig
import torch.nn.functional as F
import matplotlib.pyplot as plt
def pick_best_gpu_id():
# pick the GPU with the most free memory:
gpu_ids = [i for i in range(torch.cuda.device_count())]
print(f"# of visible GPUs: {len(gpu_ids)}")
gpu_mem = []
for gpu_id in gpu_ids:
free_memory, tot_mem = torch.cuda.mem_get_info(device=gpu_id)
gpu_mem.append(free_memory)
print("GPU %d: %d MB free" %(gpu_id, free_memory / 1024 / 1024))
if len(gpu_ids) == 0:
# no GPUs available, use CPU:
os.environ["CUDA_VISIBLE_DEVICES"] = ""
return None
best_gpu_id = gpu_ids[np.argmax(gpu_mem)]
# set this to be the active GPU:
os.environ["CUDA_VISIBLE_DEVICES"] = str(best_gpu_id)
print("Using GPU %d" %best_gpu_id)
return best_gpu_id
def seed_everything(seed: int):
random.seed(seed)
np.random.seed(seed)
+3 -1
View File
@@ -20,7 +20,8 @@ from trainer.dataset_and_utils import (
plot_grad_norms,
plot_loss,
plot_lrs,
seed_everything
seed_everything,
pick_best_gpu_id
)
from trainer.utils.lora import (
save_lora,
@@ -67,6 +68,7 @@ def main(
config: TrainingConfig,
):
seed_everything(config.seed)
pick_best_gpu_id()
(config.concept_mode, config.left_right_flip_augmentation, config.mask_target_prompts, config.clipseg_temperature, config.l1_penalty
) = modify_args_based_on_concept_mode(
-8
View File
@@ -27,14 +27,6 @@
"debug": true,
"hard_pivot": false,
"mixed_precision": "bf16",
"n_tokens": 2,
"inserting_list_tokens": [
"<s0>",
"<s1>"
],
"token_dict": {
"TOK": "<s0><s1>"
},
"unet_learning_rate": 1.0,
"lr_scheduler": "constant",
"lr_warmup_steps": 50,