diff --git a/cog_test_train.sh b/cog_test_train.sh index 8b9bf6d..9b9ef97 100644 --- a/cog_test_train.sh +++ b/cog_test_train.sh @@ -7,6 +7,5 @@ cog predict --gpus $GPU_ID \ -i concept_mode="face" \ -i sd_model_version="sdxl" \ -i max_train_steps="300" \ - -i caption_model="florence" \ -i debug="False" \ -i seed="0" \ No newline at end of file diff --git a/predict.py b/predict.py index 580358b..1df4b08 100755 --- a/predict.py +++ b/predict.py @@ -87,11 +87,6 @@ class Predictor(BasePredictor): description="Rank of LoRA embeddings for the unet.", default=16 ), - caption_model: str = Input( - description="Which model to use for captioning the images", - choices=["blip", "florence"], - default="blip" - ), n_tokens: int = Input( description="How many new tokens to train (highly recommended to leave this at 2)", ge=1, le=4, default=3 @@ -144,7 +139,7 @@ class Predictor(BasePredictor): ti_lr=ti_lr, unet_lr=unet_lr, lora_rank=lora_rank, - caption_model=caption_model, + caption_model="blip", n_tokens=n_tokens, verbose=True, debug=debug, diff --git a/requirements.txt b/requirements.txt index 81d1b40..aa5170a 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,6 +1,6 @@ -torch==2.2.1 -torchaudio==2.2.1 -torchvision==0.17.1 +torch==2.1.0 +torchaudio==2.1.0 +torchvision==0.16.0 transformers==4.38.0 diffusers==0.26.0 tokenizers==0.15.2