diff --git a/cogvideox/api/api.py b/cogvideox/api/api.py index 60c80bb..5ec9395 100644 --- a/cogvideox/api/api.py +++ b/cogvideox/api/api.py @@ -77,7 +77,7 @@ def infer_forward_api(_: gr.Blocks, app: FastAPI, controller): lora_model_path = datas.get('lora_model_path', 'none') lora_alpha_slider = datas.get('lora_alpha_slider', 0.55) prompt_textbox = datas.get('prompt_textbox', None) - negative_prompt_textbox = datas.get('negative_prompt_textbox', 'The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion.') + negative_prompt_textbox = datas.get('negative_prompt_textbox', 'The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. ') sampler_dropdown = datas.get('sampler_dropdown', 'Euler') sample_step_slider = datas.get('sample_step_slider', 30) resize_method = datas.get('resize_method', "Generate by") diff --git a/cogvideox/api/post_infer.py b/cogvideox/api/post_infer.py index 41287f5..bf7e3e5 100644 --- a/cogvideox/api/post_infer.py +++ b/cogvideox/api/post_infer.py @@ -32,10 +32,10 @@ def post_infer(generation_method, length_slider, url='http://127.0.0.1:7860'): "motion_module_path": "none", "lora_model_path": "none", "lora_alpha_slider": 0.55, - "prompt_textbox": "This video shows Mount saint helens, washington - the stunning scenery of a rocky mountains during golden hours - wide shot. A soaring drone footage captures the majestic beauty of a coastal cliff, its red and yellow stratified rock faces rich in color and against the vibrant turquoise of the sea.", - "negative_prompt_textbox": "Strange motion trajectory, a poor composition and deformed video, worst quality, normal quality, low quality, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera", + "prompt_textbox": "A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.", + "negative_prompt_textbox": "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. ", "sampler_dropdown": "Euler", - "sample_step_slider": 30, + "sample_step_slider": 50, "width_slider": 672, "height_slider": 384, "generation_method": "Video Generation", @@ -55,23 +55,16 @@ if __name__ == '__main__': # -------------------------- # # Step 1: update edition # -------------------------- # - edition = "v3" - outputs = post_update_edition(edition) - print('Output update edition: ', outputs) - - # -------------------------- # - # Step 2: update edition - # -------------------------- # - diffusion_transformer_path = "models/Diffusion_Transformer/cogvideox_funV3-XL-2-512x512" + diffusion_transformer_path = "models/Diffusion_Transformer/CogVideoX-Fun-2b-InP" outputs = post_diffusion_transformer(diffusion_transformer_path) print('Output update edition: ', outputs) # -------------------------- # - # Step 3: infer + # Step 2: infer # -------------------------- # # "Video Generation" and "Image Generation" generation_method = "Video Generation" - length_slider = 72 + length_slider = 49 outputs = post_infer(generation_method, length_slider) # Get decoded data diff --git a/cogvideox/ui/ui.py b/cogvideox/ui/ui.py index c0c5307..55f7f00 100644 --- a/cogvideox/ui/ui.py +++ b/cogvideox/ui/ui.py @@ -493,13 +493,13 @@ def ui(low_gpu_memory_mode, weight_dtype): ) prompt_textbox = gr.Textbox(label="Prompt (正向提示词)", lines=2, value="A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.") - negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion." ) + negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " ) with gr.Row(): with gr.Column(): with gr.Row(): sampler_dropdown = gr.Dropdown(label="Sampling method (采样器种类)", choices=list(scheduler_dict.keys()), value=list(scheduler_dict.keys())[0]) - sample_step_slider = gr.Slider(label="Sampling steps (生成步数)", value=30, minimum=10, maximum=100, step=1) + sample_step_slider = gr.Slider(label="Sampling steps (生成步数)", value=50, minimum=10, maximum=100, step=1) resize_method = gr.Radio( ["Generate by", "Resize according to Reference"], @@ -924,13 +924,13 @@ def ui_modelscope(model_name, savedir_sample, low_gpu_memory_mode, weight_dtype) ) prompt_textbox = gr.Textbox(label="Prompt (正向提示词)", lines=2, value="A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.") - negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion." ) + negative_prompt_textbox = gr.Textbox(label="Negative prompt (负向提示词)", lines=2, value="The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " ) with gr.Row(): with gr.Column(): with gr.Row(): sampler_dropdown = gr.Dropdown(label="Sampling method (采样器种类)", choices=list(scheduler_dict.keys()), value=list(scheduler_dict.keys())[0]) - sample_step_slider = gr.Slider(label="Sampling steps (生成步数)", value=20, minimum=10, maximum=30, step=1, interactive=False) + sample_step_slider = gr.Slider(label="Sampling steps (生成步数)", value=50, minimum=10, maximum=50, step=1, interactive=False) resize_method = gr.Radio( ["Generate by", "Resize according to Reference"], @@ -1258,13 +1258,13 @@ def ui_eas(model_name, savedir_sample): ) prompt_textbox = gr.Textbox(label="Prompt", lines=2, value="A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic.") - negative_prompt_textbox = gr.Textbox(label="Negative prompt", lines=2, value="The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion. " ) + negative_prompt_textbox = gr.Textbox(label="Negative prompt", lines=2, value="The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " ) with gr.Row(): with gr.Column(): with gr.Row(): sampler_dropdown = gr.Dropdown(label="Sampling method", choices=list(scheduler_dict.keys()), value=list(scheduler_dict.keys())[0]) - sample_step_slider = gr.Slider(label="Sampling steps", value=20, minimum=10, maximum=30, step=1, interactive=False) + sample_step_slider = gr.Slider(label="Sampling steps", value=50, minimum=10, maximum=50, step=1, interactive=False) resize_method = gr.Radio( ["Generate by", "Resize according to Reference"], diff --git a/comfyui/v1/cogvideoxfunv1_workflow_i2v.json b/comfyui/v1/cogvideoxfunv1_workflow_i2v.json index bf3dace..da770be 100644 --- a/comfyui/v1/cogvideoxfunv1_workflow_i2v.json +++ b/comfyui/v1/cogvideoxfunv1_workflow_i2v.json @@ -176,7 +176,7 @@ "Node name for S&R": "TextBox" }, "widgets_values": [ - "The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion." + "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " ] }, { diff --git a/comfyui/v1/cogvideoxfunv1_workflow_t2v.json b/comfyui/v1/cogvideoxfunv1_workflow_t2v.json index 0cd74b5..292cd2d 100644 --- a/comfyui/v1/cogvideoxfunv1_workflow_t2v.json +++ b/comfyui/v1/cogvideoxfunv1_workflow_t2v.json @@ -78,7 +78,7 @@ "Node name for S&R": "TextBox" }, "widgets_values": [ - "The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion." + "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " ] }, { diff --git a/comfyui/v1/cogvideoxfunv1_workflow_v2v.json b/comfyui/v1/cogvideoxfunv1_workflow_v2v.json index 010ae9b..e1dfcfc 100644 --- a/comfyui/v1/cogvideoxfunv1_workflow_v2v.json +++ b/comfyui/v1/cogvideoxfunv1_workflow_v2v.json @@ -184,7 +184,7 @@ "Node name for S&R": "TextBox" }, "widgets_values": [ - "The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion." + "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " ] }, { diff --git a/predict_i2v.py b/predict_i2v.py index 515dd85..8368151 100644 --- a/predict_i2v.py +++ b/predict_i2v.py @@ -51,11 +51,11 @@ validation_image_start = "asset/1.png" validation_image_end = None # prompts -prompt = "A dog is shaking head. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic." -negative_prompt = "The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion. " +prompt = "A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic." +negative_prompt = "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " guidance_scale = 6.0 seed = 43 -num_inference_steps = 25 +num_inference_steps = 50 lora_weight = 0.55 save_path = "samples/cogvideox-fun-videos_i2v" diff --git a/predict_t2v.py b/predict_t2v.py index ce35b60..e7433fc 100644 --- a/predict_t2v.py +++ b/predict_t2v.py @@ -42,11 +42,11 @@ fps = 8 # Use torch.float16 if GPU does not support torch.bfloat16 # ome graphics cards, such as v100, 2080ti, do not support torch.bfloat16 weight_dtype = torch.bfloat16 -prompt = "A dog is shaking head. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic." -negative_prompt = "The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion. " +prompt = "A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic." +negative_prompt = "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " guidance_scale = 6.0 seed = 43 -num_inference_steps = 25 +num_inference_steps = 50 lora_weight = 0.55 save_path = "samples/cogvideox-fun-videos-t2v" diff --git a/predict_v2v.py b/predict_v2v.py index 038dc52..ba4cb20 100644 --- a/predict_v2v.py +++ b/predict_v2v.py @@ -47,11 +47,11 @@ validation_video = "asset/03480_03600_good_scenes.mp4" denoise_strength = 0.70 # prompts -prompt = "A dog is shaking head. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic." -negative_prompt = "The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion. " +prompt = "A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic." +negative_prompt = "The video is not of a high quality, it has a low resolution. Watermark present in each frame. Strange motion trajectory. " guidance_scale = 6.0 seed = 43 -num_inference_steps = 25 +num_inference_steps = 50 lora_weight = 0.55 save_path = "samples/cogvideox-fun-videos_v2v"