From bc3bde68e6c2ceccaed9be031e24c89b681ea6f5 Mon Sep 17 00:00:00 2001 From: bubbliiiing <3323290568@qq.com> Date: Wed, 29 May 2024 00:34:04 +0800 Subject: [PATCH] complete data preprocess pipeline. --- easyanimate/video_caption/README.md | 79 ++++++++++++------- ...lity.py => compute_video_frame_quality.py} | 0 .../video_caption/convert_jsonl_to_json.py | 40 ++++++++++ .../datasets/put preprocess datasets here.txt | 0 .../filter_videos_by_motion_score.py | 55 +++++++++++++ easyanimate/video_caption/scenedetect_vcut.py | 59 ++++++++------ .../video_caption/stage_1_video_cut.sh | 11 +++ .../video_caption/stage_2_filter_data.sh | 45 +++++++---- .../video_caption/stage_3_video_caption.sh | 35 ++++++++ 9 files changed, 255 insertions(+), 69 deletions(-) rename easyanimate/video_caption/{video_frame_quality.py => compute_video_frame_quality.py} (100%) create mode 100644 easyanimate/video_caption/convert_jsonl_to_json.py create mode 100644 easyanimate/video_caption/datasets/put preprocess datasets here.txt create mode 100644 easyanimate/video_caption/filter_videos_by_motion_score.py create mode 100644 easyanimate/video_caption/stage_1_video_cut.sh create mode 100644 easyanimate/video_caption/stage_3_video_caption.sh diff --git a/easyanimate/video_caption/README.md b/easyanimate/video_caption/README.md index 866e4fc..b466799 100644 --- a/easyanimate/video_caption/README.md +++ b/easyanimate/video_caption/README.md @@ -22,45 +22,64 @@ EasyAnimate uses multi-modal LLMs to generate captions for frames extracted from # We strongly recommend using Docker unless you can properly handle the dependency between vllm with torch(cuda). ``` -## How to use +## Data preprocessing +Data preprocessing can be divided into three parts: -1. Prepare videos. +- Video cut. +- Video cleaning. +- Video caption. - The input for video caption can be a video folder or a metadata file (txt/csv/jsonl) containing the video path column. Please check `get_video_path_list` function in [utils/video_utils.py](utils/video_utils.py) for details. +The input for data preprocessing can be a video folder or a metadata file (txt/csv/jsonl) containing the video path column. Please check `get_video_path_list` function in [utils/video_utils.py](utils/video_utils.py) for details. -2. Generate frame captions. +For easier understanding, we use one data from Panda70m as an example for data preprocessing. Please download the video and push it in "datasets/panda_70m/before_vcut/" + +``` +📦 datasets/ +├── 📂 panda_70m/ +│ └── 📂 before_vcut/ +│ └── 📄 --C66yU3LjM_2.mp4 +``` + +1. Video cut + For long video cut, EasyAnimate utilizes PySceneDetect to identify scene changes within the video and performs scene cutting based on certain threshold values to ensure consistency in the themes of the video segments. After cutting, we only keep segments with lengths ranging from 3 to 10 seconds for model training. + + We have completed the parameters for ```stage_1_video_cut.sh```, so I can run it directly using the command sh ```stage_1_video_cut.sh```. After executing ```stage_1_video_cut.sh```, we obtained short videos in ```easyanimate/video_caption/datasets/panda_70m/train```. + + ```shell + sh stage_1_video_cut.sh + ``` +2. Video cleaning + Following SVD's data preparation process, EasyAnimate provides a simple yet effective data processing pipeline for high-quality data filtering and labeling. It also supports distributed processing to accelerate the speed of data preprocessing. The overall process is as follows: + + - Duration filtering: Analyze the basic information of the video to filter out low-quality videos that are short in duration or low in resolution. This filtering result is corresponding to the video cut (3s ~ 10s videos). + - Aesthetic filtering: Filter out videos with poor content (blurry, dim, etc.) by calculating the average aesthetic score of uniformly distributed 4 frames. + - Text filtering: Use easyocr to calculate the text proportion of middle frames to filter out videos with a large proportion of text. + - Motion filtering: Calculate interframe optical flow differences to filter out videos that move too slowly or too quickly. + + The process file of **Aesthetic filtering** is ```compute_video_frame_quality.py```. After executing ```compute_video_frame_quality.py```, we obtained the file ```datasets/panda_70m/aesthetic_score.jsonl```, where each line corresponds to the aesthetic score of each video. + + The process file of **Text filtering** is ```compute_text_score.py```. After executing ```compute_text_score.py```, we obtained the file ```datasets/panda_70m/text_score.jsonl```, where each line corresponds to the text score of each video. + + The process file of **Motion filtering** is ```compute_motion_score.py```. Motion filtering is based on Aesthetic filtering and Text filtering; only samples that meet certain aesthetic scores and text scores will undergo calculation for the Motion score. After executing ```compute_motion_score.py```, we obtained the file ```datasets/panda_70m/motion_score.jsonl```, where each line corresponds to the motion score of each video. + + Then we need to filter videos by motion scores. After executing ```filter_videos_by_motion_score.py```, we get the file ```datasets/panda_70m/train.jsonl```, which includes the video we need to caption. + + We have completed the parameters for stage_2_filter_data.sh, so I can run it directly using the command sh stage_2_filter_data.sh. + + ```shell + sh stage_2_filter_data.sh + ``` +3. Video caption + Video captioning is carried out in two stages. The first stage involves extracting frames from a video and generating descriptions for them. Subsequently, a large language model is used to summarize these descriptions into a caption. We have conducted a detailed and manual comparison of open sourced multi-modal LLMs such as [Qwen-VL](https://huggingface.co/Qwen/Qwen-VL), [ShareGPT4V-7B](https://huggingface.co/Lin-Chen/ShareGPT4V-7B), [deepseek-vl-7b-chat](https://huggingface.co/deepseek-ai/deepseek-vl-7b-chat) and etc. And we found that [llava-v1.6-vicuna-7b](https://huggingface.co/liuhaotian/llava-v1.6-vicuna-7b) is capable of generating more detailed captions with fewer hallucinations. Additionally, it is supported by serving engines like [sglang](https://github.com/sgl-project/sglang) and [lmdepoly](https://github.com/InternLM/lmdeploy), enabling faster inference. - ```shell - CUDA_VISIBLE_DEVICES=0 python caption_video_frame.py \ - --video_folder="your-video-folder/" - --frame_sample_method="mid" \ - --num_sampled_frames=1 \ - --image_caption_model_name="llava-v1.6-vicuna-7b" \ - --image_caption_prompt="Please describe this image in detail." \ - --saved_path="video_frame_caption.jsonl" - ``` + Firstly, we use ```caption_video_frame.py``` to generate frame captions. Then, we use ```caption_summary.py``` to generate summary captions. - If you cannot access to Huggingface, you can run `export HF_ENDPOINT=https://hf-mirror.com` before the above command to download the image caption model automatically. - -3. Summary frame captions. + We have completed the parameters for stage_3_video_caption.sh, so I can run it directly using the command sh stage_3_video_caption.sh. After executing ```stage_3_video_cut.sh```, we obtained last json ```train_panda_70m.json``` for easyanimate training. ```shell - CUDA_VISIBLE_DEVICES=0 python caption_summary.py \ - --video_metadata_path="video_frame_caption_result.jsonl" \ - --video_path_column="video_path" \ - --caption_column="sampled_frame_caption" \ - --summary_model_name="mistralai/Mistral-7B-Instruct-v0.2" \ - --summary_prompt="You are a helpful video description generator. I'll give you a description of the middle frame of the video clip, \ - which you need to summarize it into a description of the video clip. \ - Please provide your video description following these requirements: \ - 1. Describe the basic and necessary information of the video in the third person, be as concise as possible. \ - 2. Output the video description directly. Begin with 'In this video'. \ - 3. Limit the video description within 100 words. \ - Here is the mid-frame description: " \ - --output_dir="tmp" \ - --saved_path="video_summary_caption.jsonl" + sh stage_3_video_caption.sh ``` If you cannot access to Huggingface, you can run `export HF_ENDPOINT=https://hf-mirror.com` before the above command to download the summary caption model automatically. \ No newline at end of file diff --git a/easyanimate/video_caption/video_frame_quality.py b/easyanimate/video_caption/compute_video_frame_quality.py similarity index 100% rename from easyanimate/video_caption/video_frame_quality.py rename to easyanimate/video_caption/compute_video_frame_quality.py diff --git a/easyanimate/video_caption/convert_jsonl_to_json.py b/easyanimate/video_caption/convert_jsonl_to_json.py new file mode 100644 index 0000000..78b7ad9 --- /dev/null +++ b/easyanimate/video_caption/convert_jsonl_to_json.py @@ -0,0 +1,40 @@ +import argparse +import json +import os + +def parse_args(): + parser = argparse.ArgumentParser(description="Convert jsonl to json.") + parser.add_argument("--video_folder", type=str, default="", help="The video folder.") + parser.add_argument( + "--jsonl_load_path", type=str, default=None, help="The path to the video dataset metadata (csv/jsonl)." + ) + parser.add_argument("--save_path", type=str, default=None, help="The save path to the output results.") + args = parser.parse_args() + return args + +def main(): + args = parse_args() + + with open(args.jsonl_load_path, "r") as read: + _lines = read.readlines() + + output = [] + for line in _lines: + try: + line = json.loads(line.strip()) + videoid, name = line['video_path'], line['summary_caption'] + output.append( + { + "file_path": os.path.join(args.video_folder, videoid), + "text": name, + "type": "video", + } + ) + except: + pass + + with open(args.save_path, mode="w", encoding="utf-8") as f: + json.dump(output, f, indent=2) + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/easyanimate/video_caption/datasets/put preprocess datasets here.txt b/easyanimate/video_caption/datasets/put preprocess datasets here.txt new file mode 100644 index 0000000..e69de29 diff --git a/easyanimate/video_caption/filter_videos_by_motion_score.py b/easyanimate/video_caption/filter_videos_by_motion_score.py new file mode 100644 index 0000000..e622aaa --- /dev/null +++ b/easyanimate/video_caption/filter_videos_by_motion_score.py @@ -0,0 +1,55 @@ +import ast +import argparse +import gc +import os +from contextlib import contextmanager +from pathlib import Path + +import cv2 +import numpy as np +import pandas as pd +from joblib import Parallel, delayed +from natsort import natsorted +from tqdm import tqdm + +from utils.logger import logger +from utils.video_utils import get_video_path_list + +def parse_args(): + parser = argparse.ArgumentParser(description="Filter the motion score of the videos.") + parser.add_argument( + "--motion_score_metadata_path", type=str, default=None, help="The path to the video dataset metadata (csv/jsonl)." + ) + parser.add_argument("--low_motion_score_threshold", type=float, default=3.0, help="The low motion score threshold.") + parser.add_argument("--high_motion_score_threshold", type=float, default=8.0, help="The high motion score threshold.") + parser.add_argument("--saved_path", type=str, required=True, help="The save path to the output results (csv/jsonl).") + + args = parser.parse_args() + return args + + +def main(): + args = parse_args() + + if not (args.saved_path.endswith(".csv") or args.saved_path.endswith(".jsonl")): + raise ValueError("The saved_path must end with .csv or .jsonl.") + + if args.motion_score_metadata_path is not None: + if args.motion_score_metadata_path.endswith(".csv"): + motion_score_df = pd.read_csv(args.motion_score_metadata_path) + elif args.motion_score_metadata_path.endswith(".jsonl"): + motion_score_df = pd.read_json(args.motion_score_metadata_path, lines=True) + + filtered_motion_score_df = motion_score_df[motion_score_df["motion_score"] > args.low_motion_score_threshold] + filtered_motion_score_df = filtered_motion_score_df[motion_score_df["motion_score"] < args.high_motion_score_threshold] + + if args.saved_path.endswith(".csv"): + header = False if os.path.exists(args.saved_path) else True + filtered_motion_score_df.to_csv(args.saved_path, header=header, index=False, mode="a") + elif args.saved_path.endswith(".jsonl"): + filtered_motion_score_df.to_json(args.saved_path, orient="records", lines=True, mode="a") + logger.info(f"Save result to {args.saved_path}.") + + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/easyanimate/video_caption/scenedetect_vcut.py b/easyanimate/video_caption/scenedetect_vcut.py index 4232a52..b49c80b 100644 --- a/easyanimate/video_caption/scenedetect_vcut.py +++ b/easyanimate/video_caption/scenedetect_vcut.py @@ -1,20 +1,19 @@ import argparse -import os -from tqdm import tqdm import copy import json +import os import shutil from multiprocessing import Pool -from scenedetect import open_video, SceneManager +from scenedetect import SceneManager, open_video from scenedetect.detectors import ContentDetector from scenedetect.video_splitter import split_video_ffmpeg +from tqdm import tqdm -from utils.util import download_video, get_video_list - +from utils.video_utils import download_video, get_video_path_list tmp_file_dir = "./tmp" - +DEFAULT_FFMPEG_ARGS = '-c:v libx264 -preset veryfast -crf 22 -c:a aac' def parse_args(): parser = argparse.ArgumentParser( @@ -68,6 +67,9 @@ def parse_args(): type = int, default = os.cpu_count() // 2, help = 'Number of CPU cores to process the video scene cut.') + parser.add_argument( + "--save_json", action="store_true", help="Whether save json in datasets." + ) args = parser.parse_args() return args @@ -79,7 +81,8 @@ def split_video_into_scenes( min_seconds: int = 3, max_seconds: int = 8, save_dir: str = "", - name_template: str = "$VIDEO_NAME-Scene-$SCENE_NUMBER.mp4"): + name_template: str = "$VIDEO_NAME-Scene-$SCENE_NUMBER.mp4", + save_json: bool = False ): # SceneDetect video through casceded (threshold, FPS) frame_points = [] frame_timecode = {} @@ -105,6 +108,16 @@ def split_video_into_scenes( frame_points = sorted(frame_points) output_scene_list = [] + # Detect No Scene Change + if len(frame_points) == 0: + video = open_video(video_path, backend='pyav') + frame_points = [0, video.duration.get_frames() - 1] + frame_timecode = { + frame_points[0]: video.base_timecode, + frame_points[-1]: video.base_timecode + video.base_timecode + video.duration + } + del video + for idx in range(len(frame_points) - 1): # Limit save out min seconds if frame_points[idx+1] - frame_points[idx] < fps * min_seconds: @@ -136,33 +149,31 @@ def split_video_into_scenes( # Ensure save dir exists elif not os.path.isdir(save_dir): os.makedirs(save_dir) + clip_info_path = os.path.join(save_dir, os.path.splitext(os.path.basename(video_path))[0] + '.json') - # Early exit (If the video cannot be split into multiple scenes, save the original video). - if len(output_scene_list) == 0: - shutil.copy2(video_path, save_dir) - with open(clip_info_path, 'w', encoding='utf-8') as f: - json.dump({}, f) - return clip_info_path output_file_template = os.path.join(save_dir, name_template) split_video_ffmpeg( video_path, output_scene_list, + arg_override=DEFAULT_FFMPEG_ARGS, output_file_template=output_file_template, show_progress=False, show_output=False) # ffmpeg print - # Save clip info - json.dump( - [(frame_timecode_tuple[0].get_timecode(), frame_timecode_tuple[1].get_timecode()) for frame_timecode_tuple in output_scene_list], - open(clip_info_path, 'w'), - indent=2 - ) + + if save_json: + # Save clip info + json.dump( + [(frame_timecode_tuple[0].get_timecode(), frame_timecode_tuple[1].get_timecode()) for frame_timecode_tuple in output_scene_list], + open(clip_info_path, 'w'), + indent=2 + ) return clip_info_path def process_single_video(args): - video, threshold, frame_skip, min_seconds, max_seconds, save_dir, name_template = args + video, threshold, frame_skip, min_seconds, max_seconds, save_dir, name_template, save_json = args basename = os.path.splitext(os.path.basename(video))[0] # Video URL if video.startswith("http"): @@ -185,7 +196,8 @@ def process_single_video(args): min_seconds=min_seconds, max_seconds=max_seconds, save_dir=save_dir, - name_template=name_template + name_template=name_template, + save_json=save_json ) except Exception as e: print(e, video) @@ -202,13 +214,14 @@ def main(): save_dir = args.save_dir name_template = args.name_template num_processes = args.num_processes + save_json = args.save_json assert len(threshold) == len(frame_skip), \ "Threshold must one-to-one match frame_skip." - video_list = get_video_list(video_input) + video_list = get_video_path_list(video_input) args_list = [ - (video, threshold, frame_skip, min_seconds, max_seconds, save_dir, name_template) + (video, threshold, frame_skip, min_seconds, max_seconds, save_dir, name_template, save_json) for video in video_list ] diff --git a/easyanimate/video_caption/stage_1_video_cut.sh b/easyanimate/video_caption/stage_1_video_cut.sh new file mode 100644 index 0000000..817f314 --- /dev/null +++ b/easyanimate/video_caption/stage_1_video_cut.sh @@ -0,0 +1,11 @@ +export VIDEO_FOLDER="datasets/panda_70m/before_vcut/" +export OUTPUT_FOLDER="datasets/panda_70m/train/" + +# Cut raw videos +python scenedetect_vcut.py \ + $VIDEO_FOLDER \ + --threshold 10 20 30 \ + --frame_skip 0 1 2 \ + --min_seconds 3 \ + --max_seconds 10 \ + --save_dir $OUTPUT_FOLDER \ No newline at end of file diff --git a/easyanimate/video_caption/stage_2_filter_data.sh b/easyanimate/video_caption/stage_2_filter_data.sh index 7371621..b3d9dc4 100644 --- a/easyanimate/video_caption/stage_2_filter_data.sh +++ b/easyanimate/video_caption/stage_2_filter_data.sh @@ -1,26 +1,39 @@ -CUDA_VISIBLE_DEVICES="4,5,6,7" accelerate launch video_frame_quality.py \ - --video_metadata_path=/mnt_wg/huangkunzhe.hkz/dataset/shot2story/videos_shots/meta_file_info.jsonl \ - --video_folder=/mnt_wg/huangkunzhe.hkz/dataset/shot2story/videos_shots/data/ \ - --video_path_column=video_path \ - --metrics=AestheticScore \ +export VIDEO_FOLDER="datasets/panda_70m/train" +export FRAME_QUALITY_SAVE_PATH="datasets/panda_70m/aesthetic_score.jsonl" +export TEXT_SCORE_SAVE_PATH="datasets/panda_70m/text_score.jsonl" +export MOTION_SCORE_SAVE_PATH="datasets/panda_70m/motion_score.jsonl" +export FILTER_BY_MOTION_SCORE_SAVE_PATH="datasets/panda_70m/train.jsonl" + +# Get asethetic score of all videos +CUDA_VISIBLE_DEVICES="0" accelerate launch compute_video_frame_quality.py \ + --video_folder=$VIDEO_FOLDER \ + --video_path_column="video_path" \ + --metrics="AestheticScore" \ --saved_freq=10 \ - --saved_path=/mnt/nas/huangkunzhe.hkz/code/EasyAnimate/easyanimate/video_caption/test/aesthetic_score_shot2story.jsonl \ + --saved_path=$FRAME_QUALITY_SAVE_PATH \ --batch_size=8 -CUDA_VISIBLE_DEVICES="4,5,6,7" accelerate launch compute_text_score.py \ - --video_metadata_path=/mnt_wg/huangkunzhe.hkz/dataset/shot2story/videos_shots/meta_file_info.jsonl \ - --video_folder=/mnt_wg/huangkunzhe.hkz/dataset/shot2story/videos_shots/data/ \ +# Get text score of all videos +CUDA_VISIBLE_DEVICES="0" accelerate launch compute_text_score.py \ + --video_folder=$VIDEO_FOLDER \ --video_path_column="video_path" \ --saved_freq=10 \ - --saved_path=/mnt/nas/huangkunzhe.hkz/code/EasyAnimate/easyanimate/video_caption/test/text_score_shot2story.jsonl \ - --asethetic_score_metadata_path /mnt/nas/huangkunzhe.hkz/code/EasyAnimate/easyanimate/video_caption/test/aesthetic_score_shot2story.jsonl + --saved_path=$TEXT_SCORE_SAVE_PATH \ + --asethetic_score_metadata_path $FRAME_QUALITY_SAVE_PATH +# Get motion score after filter videos by asethetic score and text score python compute_motion_score.py \ - --video_metadata_path=/mnt_wg/huangkunzhe.hkz/dataset/shot2story/videos_shots/meta_file_info.jsonl \ - --video_folder=/mnt_wg/huangkunzhe.hkz/dataset/shot2story/videos_shots/data/ \ + --video_folder=$VIDEO_FOLDER \ --video_path_column="video_path" \ --saved_freq=10 \ - --saved_path=/mnt/nas/huangkunzhe.hkz/code/EasyAnimate/easyanimate/video_caption/test/motion_score_shot2story.jsonl \ + --saved_path=$MOTION_SCORE_SAVE_PATH \ --n_jobs=8 \ - --asethetic_score_metadata_path /mnt/nas/huangkunzhe.hkz/code/EasyAnimate/easyanimate/video_caption/test/aesthetic_score_shot2story.jsonl \ - --text_score_metadata_path /mnt/nas/huangkunzhe.hkz/code/EasyAnimate/easyanimate/video_caption/test/text_score_shot2story.jsonl \ No newline at end of file + --asethetic_score_metadata_path $FRAME_QUALITY_SAVE_PATH \ + --text_score_metadata_path $TEXT_SCORE_SAVE_PATH + +# Filter videos by motion score +python filter_videos_by_motion_score.py \ + --motion_score_metadata_path $MOTION_SCORE_SAVE_PATH \ + --low_motion_score_threshold=3 \ + --high_motion_score_threshold=8 \ + --saved_path $FILTER_BY_MOTION_SCORE_SAVE_PATH diff --git a/easyanimate/video_caption/stage_3_video_caption.sh b/easyanimate/video_caption/stage_3_video_caption.sh new file mode 100644 index 0000000..68bb0a8 --- /dev/null +++ b/easyanimate/video_caption/stage_3_video_caption.sh @@ -0,0 +1,35 @@ +export CUDA_VISIBLE_DEVICES=0 +export VIDEO_FOLDER="datasets/panda_70m/train/" +export MOTION_SCORE_META_PATH="datasets/panda_70m/train.jsonl" +export VIDEO_FRAME_CAPTION_PATH="datasets/panda_70m/frame_caption.jsonl" +export VIDEO_CAPTION_PATH="datasets/panda_70m/summary_caption.jsonl" +export LAST_JSON_PATH="datasets/panda_70m/train_panda_70m.json" + +CUDA_VISIBLE_DEVICES="0" python caption_video_frame.py \ + --video_metadata_path=$MOTION_SCORE_META_PATH \ + --video_folder=$VIDEO_FOLDER \ + --frame_sample_method="mid" \ + --num_sampled_frames=1 \ + --image_caption_model_name="llava-v1.6-vicuna-7b" \ + --image_caption_prompt="Please describe this image in detail." \ + --saved_path=$VIDEO_FRAME_CAPTION_PATH \ + --output_dir="tmp" + +CUDA_VISIBLE_DEVICES="0" python caption_summary.py \ + --video_metadata_path=$VIDEO_FRAME_CAPTION_PATH \ + --video_path_column="video_path" \ + --caption_column="sampled_frame_caption" \ + --summary_model_name="Qwen/Qwen1.5-7B-Chat" \ + --summary_prompt="You are a helpful video description generator. I'll give you a description of the middle frame of the video clip, \ + which you need to summarize it into a description of the video clip. \ + Please provide your video description following these requirements: \ + 1. Describe the basic and necessary information of the video in the third person, be as concise as possible. \ + 2. Output the video description directly. Begin with 'In this video'. \ + 3. Limit the video description within 100 words. \ + Here is the mid-frame description: " \ + --saved_path=$VIDEO_CAPTION_PATH + +python convert_jsonl_to_json.py \ + --video_folder=$VIDEO_FOLDER \ + --jsonl_load_path=$VIDEO_CAPTION_PATH \ + --save_path=$LAST_JSON_PATH \ No newline at end of file