From 90feb0709cc7cc0b2a8a0eeb574fc7e04b8e0d1f Mon Sep 17 00:00:00 2001 From: aiXander Date: Sat, 30 Mar 2024 20:37:01 -0700 Subject: [PATCH] add auto-gpu picking --- README.md | 5 ++--- test_inference.py | 16 ++++++---------- trainer/config.py | 6 +++--- trainer/dataset_and_utils.py | 21 +++++++++++++++++++++ trainer_pti.py | 4 +++- training_args.json | 8 -------- 6 files changed, 35 insertions(+), 25 deletions(-) diff --git a/README.md b/README.md index e786e0f..3d2ec1e 100755 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Code / Cleanup: - Modularize the logic in train.py as much as possible, trying to minimize dev work that needs to happen when SD3 drops (in progress) - ~~make a clean train.py entrypoint that can be run as a normal python command (instead of having to use cog)~~ - ~~make it so the textual_inversion optimizer only optimizes the actual trained token embeddings instead of all of them + resetting later~~ -- test if the trained concepts with peft are compatible with ComfyUI / AUTO1111 +- Make sure the trained concepts (with peft) are compatible with ComfyUI / AUTO1111 Algo: - Add aspect_ratio bucketing into the dataloader so we can train on non-square images (take this from https://github.com/kohya-ss/sd-scripts) @@ -30,7 +30,6 @@ Algo: - the random initialization of the token embeddings has a relatively large impact on the final outcome, there are prob ways to reduce this random variance, eg CLIP_similarity pretraining. - Improve the img captioning by swapping BLIP for cogVLM: https://github.com/THUDM/CogVLM -- add gradient clipping, see https://github.com/cloneofsimo/lora/blob/master/lora_diffusion/cli_lora_pti.py#L606C13-L608C14 Bugfixing: see msgs at: https://discord.com/channels/573691888050241543/1184175211998883950/1217550596878373037 @@ -38,10 +37,10 @@ see msgs at: https://discord.com/channels/573691888050241543/1184175211998883950 - Try to find out why the diffusers training script works for sd15 and ours doesnt: See here: https://huggingface.co/blog/sdxl_lora_advanced_script and here: https://github.com/huggingface/diffusers/tree/main/examples/advanced_diffusion_training -- figure out how to adaptively set lora_scale at inference time using peft + diffusers? (https://github.com/huggingface/peft/blob/main/src/peft/tuners/lora/layer.py#L240) Bigger improvements: +- add stronger token regularization (eg CelebBasis spanning basis) - Add multi-token training - pre-optimize token embeddings using CLIP-similarity (cfr aesthetic gradients: https://github.com/vicgalle/stable-diffusion-aesthetic-gradients/tree/main) - implement perfusion: https://research.nvidia.com/labs/par/Perfusion/ diff --git a/test_inference.py b/test_inference.py index 950218b..c1cfc90 100644 --- a/test_inference.py +++ b/test_inference.py @@ -3,7 +3,7 @@ from trainer.utils.lora import patch_pipe_with_lora, blend_conditions from trainer.utils.val_prompts import val_prompts from trainer.utils.prompt import prepare_prompt_for_lora from trainer.utils.io import make_validation_img_grid -from trainer.dataset_and_utils import seed_everything +from trainer.dataset_and_utils import seed_everything, pick_best_gpu_id from diffusers import EulerDiscreteScheduler import torch @@ -11,23 +11,18 @@ from huggingface_hub import hf_hub_download import os, json, random, time - - if __name__ == "__main__": pretrained_model = pretrained_models['sdxl'] - lora_path = 'lora_models/clipx_tiny_test---sdxl_style_lora/checkpoints/checkpoint-500' - lora_scale = 0.75 + lora_path = 'lora_models/gene---sdxl_face_lora/checkpoints/checkpoint-600' + lora_scale = 0.7 modulate_token_strength = True seed = 0 render_size = (1024, 1024) # W,H n_imgs = 4 - - - - - + n_steps = 30 + guidance_scale = 8 use_lightning = False @@ -37,6 +32,7 @@ if __name__ == "__main__": os.makedirs(output_dir, exist_ok=True) seed_everything(seed) + pick_best_gpu_id() (pipe, tokenizer_one, diff --git a/trainer/config.py b/trainer/config.py index 743203f..14a98fd 100644 --- a/trainer/config.py +++ b/trainer/config.py @@ -31,7 +31,6 @@ class TrainingConfig(BaseModel): caption_model: Literal["gpt4-v", "blip"] = "blip" left_right_flip_augmentation: bool = True augment_imgs_up_to_n: int = 20 - n_tokens: int = 2 mask_target_prompts: Union[None, str] = None crop_based_on_salience: bool = True use_face_detection_instead: bool = False @@ -43,8 +42,9 @@ class TrainingConfig(BaseModel): off_ratio_power: float = False allow_tf32: bool = True mixed_precision: Literal["fp16", "bf16", "fp32"] = "bf16" - inserting_list_tokens: List[str] = [""] - token_dict: dict = {"TOKEN": ""} + n_tokens: int = 2 + inserting_list_tokens: List[str] = ["",""] + token_dict: dict = {"TOKEN": ""} device: str = "cuda:0" crops_coords_top_left_h: int = 0 crops_coords_top_left_w: int = 0 diff --git a/trainer/dataset_and_utils.py b/trainer/dataset_and_utils.py index 941d99f..d7312c8 100755 --- a/trainer/dataset_and_utils.py +++ b/trainer/dataset_and_utils.py @@ -17,6 +17,27 @@ from transformers import AutoTokenizer, PretrainedConfig import torch.nn.functional as F import matplotlib.pyplot as plt +def pick_best_gpu_id(): + # pick the GPU with the most free memory: + gpu_ids = [i for i in range(torch.cuda.device_count())] + print(f"# of visible GPUs: {len(gpu_ids)}") + gpu_mem = [] + for gpu_id in gpu_ids: + free_memory, tot_mem = torch.cuda.mem_get_info(device=gpu_id) + gpu_mem.append(free_memory) + print("GPU %d: %d MB free" %(gpu_id, free_memory / 1024 / 1024)) + + if len(gpu_ids) == 0: + # no GPUs available, use CPU: + os.environ["CUDA_VISIBLE_DEVICES"] = "" + return None + + best_gpu_id = gpu_ids[np.argmax(gpu_mem)] + # set this to be the active GPU: + os.environ["CUDA_VISIBLE_DEVICES"] = str(best_gpu_id) + print("Using GPU %d" %best_gpu_id) + return best_gpu_id + def seed_everything(seed: int): random.seed(seed) np.random.seed(seed) diff --git a/trainer_pti.py b/trainer_pti.py index 8912505..576180c 100755 --- a/trainer_pti.py +++ b/trainer_pti.py @@ -20,7 +20,8 @@ from trainer.dataset_and_utils import ( plot_grad_norms, plot_loss, plot_lrs, - seed_everything + seed_everything, + pick_best_gpu_id ) from trainer.utils.lora import ( save_lora, @@ -67,6 +68,7 @@ def main( config: TrainingConfig, ): seed_everything(config.seed) + pick_best_gpu_id() (config.concept_mode, config.left_right_flip_augmentation, config.mask_target_prompts, config.clipseg_temperature, config.l1_penalty ) = modify_args_based_on_concept_mode( diff --git a/training_args.json b/training_args.json index 74cdd35..1f28152 100644 --- a/training_args.json +++ b/training_args.json @@ -27,14 +27,6 @@ "debug": true, "hard_pivot": false, "mixed_precision": "bf16", - "n_tokens": 2, - "inserting_list_tokens": [ - "", - "" - ], - "token_dict": { - "TOK": "" - }, "unet_learning_rate": 1.0, "lr_scheduler": "constant", "lr_warmup_steps": 50,