diff --git a/.gitignore b/.gitignore index c80e213..6feb4fb 100644 --- a/.gitignore +++ b/.gitignore @@ -4,4 +4,5 @@ remove cache __pycache__ *.tar -xander.sh \ No newline at end of file +xander.sh +.huggingface \ No newline at end of file diff --git a/README.md b/README.md index 7c8922d..ff99611 100755 --- a/README.md +++ b/README.md @@ -24,6 +24,7 @@ Code / Cleanup: - Modularize the logic in train.py as much as possible, trying to minimize dev work that needs to happen when SD3 drops Algo: +- add proper gradient accumulation so we can train with smaller bs if needed - Improve the img captioning by swapping BLIP for cogVLM: https://github.com/THUDM/CogVLM - Add aspect_ratio bucketing into the dataloader so we can train on non-square images (take this from https://github.com/kohya-ss/sd-scripts) - Tryout DoRa: https://github.com/catid/dora/tree/main diff --git a/cog.yaml b/cog.yaml index 6c97fba..b12efe3 100755 --- a/cog.yaml +++ b/cog.yaml @@ -11,7 +11,6 @@ build: - "libsm6" - "libxext6" python_packages: - #- "diffusers==0.19.3" - "diffusers==0.25.1" - "peft==0.8.2" - "torch==2.0.1" diff --git a/dataset_and_utils.py b/dataset_and_utils.py index 39cd875..46362b5 100755 --- a/dataset_and_utils.py +++ b/dataset_and_utils.py @@ -672,7 +672,6 @@ class TokenEmbeddingsHandler: new_embeddings = (text_encoder.text_model.embeddings.token_embedding.weight.data[index_updates]) off_ratio = std_token_embedding / new_embeddings.std() - print(f"---> Off-ratio[{idx}]: {off_ratio:.4f}") return off_ratio diff --git a/predict.py b/predict.py index 819a959..3e8080e 100755 --- a/predict.py +++ b/predict.py @@ -18,6 +18,11 @@ DEBUG_MODE = False load_dotenv() +os.environ["TORCH_HOME"] = "/src/.torch" +os.environ["TRANSFORMERS_CACHE"] = "/src/.huggingface/" +os.environ["DIFFUSERS_CACHE"] = "/src/.huggingface/" +os.environ["HF_HOME"] = "/src/.huggingface/" + class CogOutput(BaseModel): files: Optional[list[cogPath]] = [] name: Optional[str] = None diff --git a/preprocess.py b/preprocess.py index e087a49..852e78e 100755 --- a/preprocess.py +++ b/preprocess.py @@ -90,7 +90,7 @@ def preprocess( shutil.rmtree(working_directory) os.makedirs(working_directory) - # clear TEMP_IN_DIR first. + # Setup directories for the training data: TEMP_IN_DIR = os.path.join(working_directory, "images_in") TEMP_OUT_DIR = os.path.join(working_directory, "images_out") @@ -101,8 +101,6 @@ def preprocess( download_and_prep_training_data(input_zip_path, TEMP_IN_DIR) - print(TEMP_IN_DIR) - n_training_imgs, trigger_text, segmentation_prompt, captions = load_and_save_masks_and_captions( concept_mode, files=TEMP_IN_DIR, diff --git a/trainer_pti.py b/trainer_pti.py index d952981..a83c7e1 100755 --- a/trainer_pti.py +++ b/trainer_pti.py @@ -1,4 +1,3 @@ -# Bootstrapped from Huggingface diffuser's code. import fnmatch import json import math @@ -15,7 +14,7 @@ import torch import torch.utils.checkpoint import torch.nn.functional as F -from diffusers.models.attention_processor import LoRAAttnProcessor, LoRAAttnProcessor2_0 +from diffusers.models.attention_processor import LoRAAttnProcessor2_0 from diffusers.optimization import get_scheduler from diffusers import EulerDiscreteScheduler from safetensors.torch import save_file @@ -29,9 +28,7 @@ from dataset_and_utils import ( ) from io_utils import make_validation_img_grid -from diffusers import StableDiffusionPipeline, StableDiffusionXLPipeline from safetensors.torch import load_file - import matplotlib.pyplot as plt def compute_snr(noise_scheduler, timesteps): @@ -100,8 +97,6 @@ def plot_loss(losses, save_path='losses.png'): plt.savefig(save_path) plt.close() - - def patch_pipe_with_lora(pipe, lora_path): with open(os.path.join(lora_path, "training_args.json"), "r") as f: @@ -152,12 +147,15 @@ def patch_pipe_with_lora(pipe, lora_path): tensors = load_file(unet_path) unet.load_state_dict(tensors, strict=False) + + # Load the textual_inversion token embeddings into the pipeline: try: #SDXL handler = TokenEmbeddingsHandler([pipe.text_encoder, pipe.text_encoder_2], [pipe.tokenizer, pipe.tokenizer_2]) except: #SD15 handler = TokenEmbeddingsHandler([pipe.text_encoder, None], [pipe.tokenizer, None]) handler.load_embeddings(os.path.join(lora_path, f"{concept_name}_embeddings.safetensors")) + return pipe def get_avg_lr(optimizer): @@ -223,7 +221,6 @@ def prepare_prompt_for_lora(prompt, lora_path, interpolation=False, verbose=True except: # fallback for old loras that dont have the name field: return training_args["trigger_text"] + ", " + prompt - print(f"lora name: {lora_name}") lora_name_encapsulated = "<" + lora_name + ">" trigger_text = training_args["trigger_text"] @@ -236,7 +233,6 @@ def prepare_prompt_for_lora(prompt, lora_path, interpolation=False, verbose=True mode = "object" # Handle different modes - print(f"lora mode: {mode}") if mode != "style": replacements = { "": trigger_text, @@ -289,9 +285,9 @@ def prepare_prompt_for_lora(prompt, lora_path, interpolation=False, verbose=True return prompt - +from val_prompts import val_prompts @torch.no_grad() -def render_images(lora_path, train_step, seed, is_lora, pretrained_model, lora_scale = 0.7, n_imgs = 4, debug = False, device = "cuda:0"): +def render_images(pipeline, lora_path, train_step, seed, is_lora, pretrained_model, lora_scale = 0.7, n_imgs = 4, debug = False, device = "cuda:0"): random.seed(seed) @@ -300,107 +296,39 @@ def render_images(lora_path, train_step, seed, is_lora, pretrained_model, lora_s concept_mode = training_args["concept_mode"] if concept_mode == "style": - validation_prompts = [ - 'a beautiful mountainous landscape, boulders, fresh water stream, setting sun', - 'the stunning skyline of New York City', - 'fruit hanging from a tree, highly detailed texture, soil, rain, drops, photo realistic, surrealism, highly detailed, 8k macrophotography', - 'the Taj Mahal, stunning wallpaper', - 'A majestic tree rooted in circuits, leaves shimmering with data streams, stands as a beacon where the digital dawn caresses the fog-laden, binary soil—a symphony of pixels and chlorophyll.', - 'A beautiful octopus, with swirling tendrils and a pulsating heart of fiery opal hues, hovers ethereally against a starry void, sculpted through a meticulous flame-working technique.', - 'a stunning image of an aston martin sportscar', - 'the streets of new york city, traffic lights, skyline, setting sun', - 'An ethereal, levitating monolith backlit by a supernova sky, casting iridescent light on the ice-spiked Martian terrain. Neo-futurism, Dali surrealism, wide-angle lens, chiaroscuro lighting.', - 'a portrait of a beautiful young woman', - 'a luminous white lotus blossom floats on rippling waters, green petals', - 'the all seeing eye made of golden feathers, surrounded by waterfall, photorealistic, ethereal aesthetics, powerful', - 'A luminescent glass butterfly, wings shimmering elegantly, depicts deftly the fragility yet adamantine spirit of nature. It encapsulates Atari honkaku technique, glowing embers inside capturing the sun as it gracefully clenches lifes sweet unpredictability', - 'Glass Roots: A luminescent glass sculpture of a fully bloomed rose emerges from a broken marble pedestal, natures resilience triumphant amidst the decay. Shadows cast by a dim overhead spotlight. Delicate veins intertwine the transparent petals, illuminating from within, symbolizing fragilitys steely core.', - 'Eternal Arbor Description: A colossal, life-size tapestry hangs majestically in a dimly lit chamber. Its profound serenity contrasts the grand spectacle it unfolds. Hundreds of intricately woven stitches meticulously portray a towering, ancient oak tree, its knotted branches embracing the heavens. ', - 'In the heart of an ancient forest, a massive projection illuminates the darkness. A lone figure, a majestic mythical creature made of shimmering gold, materializes, casting a radiant glow amidst the towering trees. intricate geometric surfaces encasing an expanse of flora and fauna,', - 'The Silent Of Silicon, a digital deer rendered in hyper-realistic 3D, eyes glowing in binary code, comfortably resting amidst rich motherboard-green foliage, accented under crisply fluorescent, simulated LED dawn.', - 'owl made up of geometric shapes, contours of glowing plasma, black background, dramatic, full picture, ultra high res, octane', - 'A twisting creature of reflective dragonglass swirling above a scorched field amidst a large clearing in a dark forest', - 'what do i say to make me exist, oriental mythical beasts, in the golden danish age, in the history of television in the style of light violet and light red, serge najjar, playful and whimsical, associated press photo, afrofuturism-inspired, alasdair mclellan, electronic media', - 'A towering, rusted iron monolith emerges from a desolate cityscape, piercing the horizon with audacious defiance. Amidst contrasting patches of verdant, natures forgotten touch yearns for connection, provoking intense introspection and tumultuous emotions. vibrant splatters of chaotic paint epitom', - 'A humanoid figure with a luminous, translucent body floats in a vast, ethereal digital landscape. Strands of brilliant, iridescent code rain down, intertwining with the figure. a blend of human features and intricate circuitry, hinting at the merging of organic and digital existence', - 'In the heart of a dense digital forest, a majestic, crystalline unicorn rises. Its translucent, pixelated mane seamlessly transitions into the vibrant greens and golds of the surrounding floating circuit board leaves. Soft moonlight filters through the gaps, creating a breathtaking', - 'Silver mushroom with gem spots emerging from water', - 'Binary Love: A heart-shaped composition made up of glowing binary code, symbolizing the merging of human emotion and technology, incredible digital art, cyberpunk, neon colors, glitch effects, 3D octane render, HD', - 'A labyrinthine maze representing the search for answers and understanding, Abstract expressionism, muted color palette, heavy brushstrokes, textured surfaces, somber atmosphere, symbolic elements', - 'A solitary tree standing tall amidst a sea of buildings, Urban nature photography, vibrant colors, juxtaposition of natural elements with urban landscapes, play of light and shadow, storytelling through compositions', - ] - validation_prompts = random.sample(validation_prompts, n_imgs) - validation_prompts[0] = '' + validation_prompts_raw = random.sample(val_prompts['style'], n_imgs) + validation_prompts_raw[0] = '' elif concept_mode == "face": - validation_prompts = [ - "an intricate wood carving of in a historic temple", - ' as pixel art, 8-bit video game style', - 'painting of by Vincent van Gogh', - ' as a superhero, wearing a cape', - ' as a statue made of marble', - ' as a character in a noir graphic novel, under a rain-soaked streetlamp', - 'stop motion animation of using clay, Wallace and Gromit style', - ' portrayed in a famous renaissance painting, replacing Mona Lisas face', - 'a photo of attending the Oscars, walking down the red carpet with sunglasses', - ' as a pop vinyl figure, complete with oversized head and small body', - ' as a retro holographic sticker, shimmering in bright colors', - ' as a bobblehead on a car dashboard, nodding incessantly', - " captured in a snow globe, complete with intricate details", - "a photo of climbing mount Everest in the snow, alpinism", - " as an action figure superhero, lego toy, toy story", - 'a photo of a massive statue of in the middle of the city', - 'a masterful oil painting portraying with vibrant colors, brushstrokes and textures', - 'a vibrant low-poly artwork of , rendered in SVG, vector graphics', - ', polaroid photograph', - 'a huge sand sculpture on a sunny beach, made of sand', - ' immortalized as an exquisite marble statue with masterful chiseling, swirling marble patterns and textures', - ] - validation_prompts = random.sample(validation_prompts, n_imgs) - validation_prompts[0] = '' + validation_prompts_raw = random.sample(val_prompts['face'], n_imgs) + validation_prompts_raw[0] = '' else: - validation_prompts = [ - "an intricate wood carving of in a historic temple", - " captured in a snow globe, complete with intricate details", - " as a retro holographic sticker, shimmering in bright colors", - "a painting of by Vincent van Gogh, impressionism, oil painting, vibrant colors, texture", - " as an action figure superhero, lego toy, toy story", - 'a photo of a massive statue of in the middle of the city', - 'a masterful oil painting portraying with vibrant colors, thick brushstrokes, abstract, surrealism', - 'an intricate origami paper sculpture of ', - 'a vibrant low-poly artwork of , rendered in SVG, vector graphics', - 'an artistic polaroid photograph of , vintage', - ' immortalized as an exquisite marble statue with masterful chiseling, swirling marble patterns and textures', - 'a colorful and dynamic mural sprawling across the side of a building in a city pulsing with life', - " transformed into a stained glass window, casting vibrant colors in the light", - "A whimsical papier-mâché sculpture of , bursting with color and whimsy.", - " as a futuristic neon sign, glowing vividly in the night, cyberpunk, vaporwave", - "A detailed graphite pencil sketch of , showcasing shadows and depth, pencil drawing, grayscale", - " reimagined as a detailed mechanical model, complete with moving parts, metal, gears, steampunk", - "A vibrant pixel art representation of , classic 8-bit video game", - "A breathtaking ice sculpture of , carved with precision and clarity, ice carving, frozen", - ] - validation_prompts = random.sample(validation_prompts, n_imgs) - validation_prompts[0] = '' + validation_prompts_raw = random.sample(val_prompts['object'], n_imgs) + validation_prompts_raw[0] = '' + # Try to free up some memory before rendering the images + gc.collect() torch.cuda.empty_cache() - print(f"Loading inference pipeline from {pretrained_model['path']}...") - (pipeline, - tokenizer_one, - tokenizer_two, - noise_scheduler, - text_encoder_one, - text_encoder_two, - vae, - unet) = load_models(pretrained_model, device, torch.float16) + if 1: # reload the entire pipeline from disk and load in the lora module + (pipeline, + tokenizer_one, + tokenizer_two, + noise_scheduler, + text_encoder_one, + text_encoder_two, + vae, + unet) = load_models(pretrained_model, device, torch.float16) + + pipeline = pipeline.to(device) + pipeline = patch_pipe_with_lora(pipeline, lora_path) + pipeline.scheduler = EulerDiscreteScheduler.from_config(pipeline.scheduler.config) + + else: + print(f"Re-using training pipeline for inference") - pipeline = pipeline.to(device) - pipeline = patch_pipe_with_lora(pipeline, lora_path) - pipeline.scheduler = EulerDiscreteScheduler.from_config(pipeline.scheduler.config) - validation_prompts_raw = validation_prompts - validation_prompts = [prepare_prompt_for_lora(prompt, lora_path) for prompt in validation_prompts] + validation_prompts = [prepare_prompt_for_lora(prompt, lora_path) for prompt in validation_prompts_raw] generator = torch.Generator(device=device).manual_seed(0) pipeline_args = { "negative_prompt": "nude, naked, poorly drawn face, ugly, tiling, out of frame, extra limbs, disfigured, deformed body, blurry, blurred, watermark, text, grainy, signature, cut off, draft", @@ -520,6 +448,8 @@ def main( elif mixed_precision == "bf16": weight_dtype = torch.bfloat16 + print(f"Loading models with weight_dtype: {weight_dtype}") + if scale_lr: unet_learning_rate = ( unet_learning_rate * gradient_accumulation_steps * train_batch_size @@ -752,18 +682,11 @@ def main( shutil.rmtree(checkpoint_dir) os.makedirs(f"{checkpoint_dir}") - # Experimental: warmup the token embeddings using CLIP-similarity: + # Experimental TODO: warmup the token embeddings using CLIP-similarity optimization #embedding_handler.pre_optimize_token_embeddings(train_dataset) ti_lrs, lora_lrs = [], [] - - # Count the total number of lora parameters - total_n_lora_params = sum(p.numel() for p in unet_lora_parameters) - - #output_save_dir = f"{checkpoint_dir}/checkpoint-{global_step}" - #save(output_save_dir, global_step, unet, embedding_handler, token_dict, args_dict, seed, is_lora, unet_param_to_optimize_names) losses = [] - start_time, images_done = time.time(), 0 for epoch in range(first_epoch, num_train_epochs): @@ -902,24 +825,9 @@ def main( if l1_penalty > 0.0: # Compute normalized L1 norm (mean of abs sum) of all lora parameters: - l1_norm = sum(p.abs().sum() for p in unet_lora_parameters) / total_n_lora_params - #print(f"Loss: {loss.item():.6f}, L1-penalty: {(l1_penalty * l1_norm).item():.6f}") + l1_norm = sum(p.abs().sum() for p in unet_lora_parameters) / sum(p.numel() for p in unet_lora_parameters) loss += l1_penalty * l1_norm - if global_step % 100 == 0 and debug: - embedding_handler.print_token_info() - - if global_step % (max_train_steps//3) == 0 and debug: - plot_torch_hist(unet_lora_parameters, global_step, output_dir, "lora_weights", min_val=-0.3, max_val=0.3, ymax_f = 0.05) - - token_embeddings = embedding_handler.get_trainable_embeddings() - for i, token_embeddings_i in enumerate(token_embeddings): - plot_torch_hist(token_embeddings_i[0], global_step, output_dir, f"embeddings_weights_token_0_{i}", min_val=-0.05, max_val=0.05, ymax_f = 0.05) - plot_torch_hist(token_embeddings_i[1], global_step, output_dir, f"embeddings_weights_token_1_{i}", min_val=-0.05, max_val=0.05, ymax_f = 0.05) - - plot_loss(losses, save_path=f'{output_dir}/losses.png') - plot_lrs(lora_lrs, ti_lrs, save_path=f'{output_dir}/learning_rates.png') - losses.append(loss.item()) loss.backward() @@ -943,14 +851,22 @@ def main( lora_lrs.append(get_avg_lr(optimizer_prod)) # Print some statistics: - if (global_step % checkpointing_steps == 0) and (global_step > 0): + if (global_step % checkpointing_steps == 0): output_save_dir = f"{checkpoint_dir}/checkpoint-{global_step}" save(output_save_dir, global_step, unet, embedding_handler, token_dict, args_dict, seed, is_lora, unet_lora_parameters, unet_param_to_optimize_names) + validation_prompts = render_images(pipe, output_save_dir, global_step, seed, is_lora, pretrained_model, n_imgs = 4, debug=debug) last_save_step = global_step + if debug: + token_embeddings = embedding_handler.get_trainable_embeddings() + for i, token_embeddings_i in enumerate(token_embeddings): + plot_torch_hist(token_embeddings_i[0], global_step, output_dir, f"embeddings_weights_token_0_{i}", min_val=-0.05, max_val=0.05, ymax_f = 0.05) + plot_torch_hist(token_embeddings_i[1], global_step, output_dir, f"embeddings_weights_token_1_{i}", min_val=-0.05, max_val=0.05, ymax_f = 0.05) + + embedding_handler.print_token_info() + plot_torch_hist(unet_lora_parameters, global_step, output_dir, "lora_weights", min_val=-0.3, max_val=0.3, ymax_f = 0.05) plot_loss(losses, save_path=f'{output_dir}/losses.png') plot_lrs(lora_lrs, ti_lrs, save_path=f'{output_dir}/learning_rates.png') - validation_prompts = render_images(output_save_dir, global_step, seed, is_lora, pretrained_model, n_imgs = 4, debug=debug) images_done += train_batch_size global_step += 1 @@ -991,7 +907,7 @@ def main( gc.collect() torch.cuda.empty_cache() - validation_prompts = render_images(output_save_dir, global_step, seed, is_lora, pretrained_model, n_imgs = 4, debug=debug) + validation_prompts = render_images(pipe, output_save_dir, global_step, seed, is_lora, pretrained_model, n_imgs = 4, debug=debug) with open(f"{output_save_dir}/training_args.json", "w") as f: args_dict["grid_prompts"] = validation_prompts diff --git a/val_prompts.py b/val_prompts.py new file mode 100644 index 0000000..bc703c5 --- /dev/null +++ b/val_prompts.py @@ -0,0 +1,77 @@ + +val_prompts = {} +val_prompts['style'] = [ + 'a beautiful mountainous landscape, boulders, fresh water stream, setting sun', + 'the stunning skyline of New York City', + 'fruit hanging from a tree, highly detailed texture, soil, rain, drops, photo realistic, surrealism, highly detailed, 8k macrophotography', + 'the Taj Mahal, stunning wallpaper', + 'A majestic tree rooted in circuits, leaves shimmering with data streams, stands as a beacon where the digital dawn caresses the fog-laden, binary soil—a symphony of pixels and chlorophyll.', + 'A beautiful octopus, with swirling tendrils and a pulsating heart of fiery opal hues, hovers ethereally against a starry void, sculpted through a meticulous flame-working technique.', + 'a stunning image of an aston martin sportscar', + 'the streets of new york city, traffic lights, skyline, setting sun', + 'An ethereal, levitating monolith backlit by a supernova sky, casting iridescent light on the ice-spiked Martian terrain. Neo-futurism, Dali surrealism, wide-angle lens, chiaroscuro lighting.', + 'a portrait of a beautiful young woman', + 'a luminous white lotus blossom floats on rippling waters, green petals', + 'the all seeing eye made of golden feathers, surrounded by waterfall, photorealistic, ethereal aesthetics, powerful', + 'A luminescent glass butterfly, wings shimmering elegantly, depicts deftly the fragility yet adamantine spirit of nature. It encapsulates Atari honkaku technique, glowing embers inside capturing the sun as it gracefully clenches lifes sweet unpredictability', + 'Glass Roots: A luminescent glass sculpture of a fully bloomed rose emerges from a broken marble pedestal, natures resilience triumphant amidst the decay. Shadows cast by a dim overhead spotlight. Delicate veins intertwine the transparent petals, illuminating from within, symbolizing fragilitys steely core.', + 'Eternal Arbor Description: A colossal, life-size tapestry hangs majestically in a dimly lit chamber. Its profound serenity contrasts the grand spectacle it unfolds. Hundreds of intricately woven stitches meticulously portray a towering, ancient oak tree, its knotted branches embracing the heavens. ', + 'In the heart of an ancient forest, a massive projection illuminates the darkness. A lone figure, a majestic mythical creature made of shimmering gold, materializes, casting a radiant glow amidst the towering trees. intricate geometric surfaces encasing an expanse of flora and fauna,', + 'The Silent Of Silicon, a digital deer rendered in hyper-realistic 3D, eyes glowing in binary code, comfortably resting amidst rich motherboard-green foliage, accented under crisply fluorescent, simulated LED dawn.', + 'owl made up of geometric shapes, contours of glowing plasma, black background, dramatic, full picture, ultra high res, octane', + 'A twisting creature of reflective dragonglass swirling above a scorched field amidst a large clearing in a dark forest', + 'what do i say to make me exist, oriental mythical beasts, in the golden danish age, in the history of television in the style of light violet and light red, serge najjar, playful and whimsical, associated press photo, afrofuturism-inspired, alasdair mclellan, electronic media', + 'A towering, rusted iron monolith emerges from a desolate cityscape, piercing the horizon with audacious defiance. Amidst contrasting patches of verdant, natures forgotten touch yearns for connection, provoking intense introspection and tumultuous emotions. vibrant splatters of chaotic paint epitom', + 'A humanoid figure with a luminous, translucent body floats in a vast, ethereal digital landscape. Strands of brilliant, iridescent code rain down, intertwining with the figure. a blend of human features and intricate circuitry, hinting at the merging of organic and digital existence', + 'In the heart of a dense digital forest, a majestic, crystalline unicorn rises. Its translucent, pixelated mane seamlessly transitions into the vibrant greens and golds of the surrounding floating circuit board leaves. Soft moonlight filters through the gaps, creating a breathtaking', + 'Silver mushroom with gem spots emerging from water', + 'Binary Love: A heart-shaped composition made up of glowing binary code, symbolizing the merging of human emotion and technology, incredible digital art, cyberpunk, neon colors, glitch effects, 3D octane render, HD', + 'A labyrinthine maze representing the search for answers and understanding, Abstract expressionism, muted color palette, heavy brushstrokes, textured surfaces, somber atmosphere, symbolic elements', + 'A solitary tree standing tall amidst a sea of buildings, Urban nature photography, vibrant colors, juxtaposition of natural elements with urban landscapes, play of light and shadow, storytelling through compositions', + ] + +val_prompts["face"] = [ + "an intricate wood carving of in a historic temple", + ' as pixel art, 8-bit video game style', + 'painting of by Vincent van Gogh', + ' as a superhero, wearing a cape', + ' as a statue made of marble', + ' as a character in a noir graphic novel, under a rain-soaked streetlamp', + 'stop motion animation of using clay, Wallace and Gromit style', + ' portrayed in a famous renaissance painting, replacing Mona Lisas face', + 'a photo of attending the Oscars, walking down the red carpet with sunglasses', + ' as a pop vinyl figure, complete with oversized head and small body', + ' as a retro holographic sticker, shimmering in bright colors', + ' as a bobblehead on a car dashboard, nodding incessantly', + " captured in a snow globe, complete with intricate details", + "a photo of climbing mount Everest in the snow, alpinism", + " as an action figure superhero, lego toy, toy story", + 'a photo of a massive statue of in the middle of the city', + 'a masterful oil painting portraying with vibrant colors, brushstrokes and textures', + 'a vibrant low-poly artwork of , rendered in SVG, vector graphics', + ', polaroid photograph', + 'a huge sand sculpture on a sunny beach, made of sand', + ' immortalized as an exquisite marble statue with masterful chiseling, swirling marble patterns and textures', + ] + +val_prompts["object"] = [ + "an intricate wood carving of in a historic temple", + " captured in a snow globe, complete with intricate details", + " as a retro holographic sticker, shimmering in bright colors", + "a painting of by Vincent van Gogh, impressionism, oil painting, vibrant colors, texture", + " as an action figure superhero, lego toy, toy story", + 'a photo of a massive statue of in the middle of the city', + 'a masterful oil painting portraying with vibrant colors, thick brushstrokes, abstract, surrealism', + 'an intricate origami paper sculpture of ', + 'a vibrant low-poly artwork of , rendered in SVG, vector graphics', + 'an artistic polaroid photograph of , vintage', + ' immortalized as an exquisite marble statue with masterful chiseling, swirling marble patterns and textures', + 'a colorful and dynamic mural sprawling across the side of a building in a city pulsing with life', + " transformed into a stained glass window, casting vibrant colors in the light", + "A whimsical papier-mâché sculpture of , bursting with color and whimsy.", + " as a futuristic neon sign, glowing vividly in the night, cyberpunk, vaporwave", + "A detailed graphite pencil sketch of , showcasing shadows and depth, pencil drawing, grayscale", + " reimagined as a detailed mechanical model, complete with moving parts, metal, gears, steampunk", + "A vibrant pixel art representation of , classic 8-bit video game", + "A breathtaking ice sculpture of , carved with precision and clarity, ice carving, frozen", + ] \ No newline at end of file