From 19a8b5385e025e1595d9b41b51c8b944ca6f3dcc Mon Sep 17 00:00:00 2001 From: Fillip Isgro Date: Thu, 13 Mar 2025 13:12:34 -0500 Subject: [PATCH] YEETED moviepy --- .DS_Store | Bin 6148 -> 6148 bytes nodes/FL_GeminiVideoCaptioner.py | 473 +++++++++++++++++-------------- 2 files changed, 261 insertions(+), 212 deletions(-) diff --git a/.DS_Store b/.DS_Store index 85272394072fd7381873ee541a44ec73f7142747..d568d1dee1d1e2352e589a33599b482536aa5449 100644 GIT binary patch delta 14 VcmZoMXffEJ!o 0 else None - + # Create WebM video from frames print(f"[FL_GeminiVideoCaptioner] Creating WebM video from {len(frames)} frames...") webm_path = self.create_webm_from_frames(frames, fps=frames_per_second) - + if webm_path is None: print(f"[FL_GeminiVideoCaptioner] WebM creation failed, falling back to frame processing...") # Generate caption from individual frames @@ -211,7 +212,7 @@ class FL_GeminiVideoCaptioner: # Use the WebM file for captioning mime_type = "video/webm" print(f"[FL_GeminiVideoCaptioner] Using WebM video for captioning") - + # Get caption using the WebM file caption = self.get_caption_with_video_file( api_key, @@ -226,29 +227,29 @@ class FL_GeminiVideoCaptioner: top_k, seed ) - + # Clean up temporary WebM file os.unlink(webm_path) - + return (caption, sample_frame_tensor) - + def get_video_info(self, video_path): """Get basic information about the video file""" cap = cv2.VideoCapture(video_path) if not cap.isOpened(): raise ValueError(f"Could not open video file: {video_path}") - + # Get video properties frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) fps = cap.get(cv2.CAP_PROP_FPS) width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) - + # Calculate duration in seconds duration = frame_count / fps if fps > 0 else 0 - + cap.release() - + return { 'frame_count': frame_count, 'fps': fps, @@ -256,62 +257,62 @@ class FL_GeminiVideoCaptioner: 'height': height, 'duration': duration } - + def extract_frames(self, video_path, target_fps, max_seconds): """Extract frames at specified fps from a video file, up to max_seconds""" cap = cv2.VideoCapture(video_path) if not cap.isOpened(): return [], [] - + # Get video properties original_fps = cap.get(cv2.CAP_PROP_FPS) frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) - + # Calculate how many frames to skip to achieve target_fps if original_fps <= 0: print("[Warning] Could not determine video FPS. Using 30 fps as default.") original_fps = 30 - + frame_interval = max(1, round(original_fps / target_fps)) - + # Calculate total duration and limit frames duration = min(frame_count / original_fps, max_seconds) max_frames = int(duration * target_fps) - + frames = [] timestamps = [] # in seconds - + progress = ProgressBar(max_frames) frame_idx = 0 frame_count = 0 - + while True: ret, frame = cap.read() if not ret or len(frames) >= max_frames: break - + # Process frame at intervals to achieve target_fps if frame_idx % frame_interval == 0: # Convert BGR to RGB frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) frames.append(frame_rgb) - + # Calculate timestamp in seconds timestamp = frame_idx / original_fps timestamps.append(timestamp) - + progress.update_absolute(frame_count) frame_count += 1 - + frame_idx += 1 - + cap.release() return frames, timestamps - + def get_gemini_caption(self, api_key, frames, timestamps, video_path, mime_type, prompt, model, process_audio, temperature, max_output_tokens, top_p, top_k, seed): """Send frames to Gemini API and get caption response""" - + # For newer models that support direct video input if model in ["gemini-1.5-pro", "gemini-1.5-flash", "gemini-2.0-flash"]: return self.get_caption_with_video_file( @@ -324,19 +325,18 @@ class FL_GeminiVideoCaptioner: api_key, frames, prompt, model, temperature, max_output_tokens, top_p, top_k, seed ) - - # This method is no longer needed as we directly call get_caption_with_frames - - def get_caption_with_frames(self, api_key, frames, prompt, model, temperature, max_output_tokens, top_p, top_k, seed): + + def get_caption_with_frames(self, api_key, frames, prompt, model, temperature, max_output_tokens, top_p, top_k, + seed): """Send individual frames to Gemini API""" # Use API version v1beta for all models api_url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={api_key}" - + content_parts = [] - + # Set frame limit based on model max_frames = 20 # Default for Gemini 1.0 - + if model.startswith("gemini-1.5") or model.startswith("gemini-2.0"): max_frames = min(len(frames), 60) # Can handle more frames for newer models # For these models, we should put the frames first, then the prompt @@ -345,26 +345,27 @@ class FL_GeminiVideoCaptioner: max_frames = min(len(frames), 20) # Limit to 20 frames for 1.0 # Put the prompt first for older models content_parts.append({"text": prompt}) - + frames_to_process = frames[:max_frames] - - print(f"[FL_GeminiVideoCaptioner] Processing {len(frames_to_process)} out of {len(frames)} total frames for {model}") + + print( + f"[FL_GeminiVideoCaptioner] Processing {len(frames_to_process)} out of {len(frames)} total frames for {model}") progress = ProgressBar(len(frames_to_process)) - + # Add frames as mimetype image/jpeg for i, frame in enumerate(frames_to_process): # Convert numpy array to PIL Image img = Image.fromarray(frame) - + # Save image to temp file with tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) as temp: img.save(temp, format="JPEG") temp_filename = temp.name - + # Read the image file and encode to base64 with open(temp_filename, "rb") as img_file: base64_image = base64.b64encode(img_file.read()).decode("utf-8") - + # Add to content parts content_parts.append({ "inline_data": { @@ -372,15 +373,15 @@ class FL_GeminiVideoCaptioner: "data": base64_image } }) - + # Delete temp file os.unlink(temp_filename) progress.update_absolute(i) - + # For Gemini 1.5+ models, add the prompt after the frames if model.startswith("gemini-1.5") or model.startswith("gemini-2.0"): content_parts.append({"text": prompt}) - + # Prepare request payload payload = { "contents": [{ @@ -395,60 +396,61 @@ class FL_GeminiVideoCaptioner: "seed": seed } } - + # Send request print(f"[FL_GeminiVideoCaptioner] Sending request to Gemini API ({model})...") return self._send_api_request(api_url, payload) - + def get_caption_with_video_file(self, api_key, video_path, mime_type, prompt, model, process_audio, temperature, max_output_tokens, top_p, top_k, seed): """Send entire video file to Gemini API (for Gemini 1.5+ models)""" # Use API version v1beta for all models api_url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={api_key}" - + # For newer models (1.5+), we can send the entire video file print(f"[FL_GeminiVideoCaptioner] Processing video file directly with {model}") - + # Check file size file_size = os.path.getsize(video_path) max_request_size = 30 * 1024 * 1024 # 30MB - + if file_size > max_request_size: - print(f"[FL_GeminiVideoCaptioner] Warning: Video file size ({file_size / (1024 * 1024):.2f}MB) exceeds the direct API request limit of 30MB.") - + print( + f"[FL_GeminiVideoCaptioner] Warning: Video file size ({file_size / (1024 * 1024):.2f}MB) exceeds the direct API request limit of 30MB.") + # Try to compress the video to WebM if not already if not mime_type or mime_type != "video/webm": print(f"[FL_GeminiVideoCaptioner] Attempting to compress to WebM format...") webm_path = self.convert_to_webm(video_path) - + if webm_path: # Use the compressed WebM file instead video_path = webm_path mime_type = "video/webm" file_size = os.path.getsize(video_path) print(f"[FL_GeminiVideoCaptioner] Compressed to WebM: {file_size / (1024 * 1024):.2f}MB") - + # If still too large, fall back to frame extraction if file_size > max_request_size: print(f"[FL_GeminiVideoCaptioner] Video still too large, falling back to frame extraction...") frames, timestamps = self.extract_frames(video_path, 1.0, 300) # 1 fps, max 5 minutes - + # Clean up temporary WebM file if we created one if 'webm_path' in locals() and webm_path: os.unlink(webm_path) - + if not frames: return "Error: Failed to extract frames from large video file" - + return self.get_caption_with_frames( api_key, frames, prompt, model, temperature, max_output_tokens, top_p, top_k, seed ) - + # Read video file and encode to base64 with open(video_path, "rb") as video_file: video_data = video_file.read() base64_video = base64.b64encode(video_data).decode("utf-8") - + # Prepare API request content_parts = [{ "inline_data": { @@ -458,7 +460,7 @@ class FL_GeminiVideoCaptioner: }, { "text": prompt }] - + # Prepare request payload payload = { "contents": [{ @@ -472,209 +474,256 @@ class FL_GeminiVideoCaptioner: "topK": top_k } } - + # Send request print(f"[FL_GeminiVideoCaptioner] Sending video to Gemini API ({model})...") return self._send_api_request(api_url, payload) - + def convert_to_webm(self, input_path, max_size_mb=29): - """Convert any video to WebM format with size limit using MoviePy - + """Convert any video to WebM format with size limit using OpenCV + Args: input_path: Path to input video file max_size_mb: Maximum size in MB for the output WebM file - + Returns: Path to the converted WebM file """ - # Create a temporary file for the output - with tempfile.NamedTemporaryFile(suffix=".webm", delete=False) as temp_webm: - output_path = temp_webm.name - + output_path = None + cap = None + out = None + try: - # Load the input video - video = VideoFileClip(input_path) - + # Create a temporary file for the output + with tempfile.NamedTemporaryFile(suffix=".webm", delete=False) as temp_webm: + output_path = temp_webm.name + + # Open input video + cap = cv2.VideoCapture(input_path) + if not cap.isOpened(): + raise ValueError(f"Could not open video file: {input_path}") + # Get video properties - duration = video.duration - fps = video.fps - width, height = video.size - + fps = cap.get(cv2.CAP_PROP_FPS) + width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) + height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) + frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) + duration = frame_count / fps if fps > 0 else 0 + # Cap duration if very long video (45 minutes max) max_duration = 45 * 60 # 45 minutes in seconds - if duration > max_duration: - print(f"[FL_GeminiVideoCaptioner] Video duration ({duration}s) exceeds 45 minutes, trimming to {max_duration}s") - video = video.subclip(0, max_duration) - duration = max_duration - + max_frames = int(min(duration, max_duration) * fps) + # Start with high quality and gradually reduce max_size_bytes = max_size_mb * 1024 * 1024 - - # Set size reduction factors and quality size_factors = [1.0, 0.75, 0.5, 0.25, 0.125] # Progressive size reduction - + for size_factor in size_factors: # Calculate new dimensions new_width = int(width * size_factor) new_height = int(height * size_factor) - - # Ensure dimensions are even (required for some codecs) + + # Ensure dimensions are even new_width = new_width if new_width % 2 == 0 else new_width + 1 new_height = new_height if new_height % 2 == 0 else new_height + 1 - - # Try different bitrates - for bitrate_factor in [1.0, 0.75, 0.5, 0.25, 0.1]: - # Estimate a reasonable bitrate based on resolution - base_bitrate = new_width * new_height * fps * 0.000005 - bitrate = int(base_bitrate * bitrate_factor) - - print(f"[FL_GeminiVideoCaptioner] Converting to WebM: {new_width}x{new_height}, bitrate={bitrate}K") - - # Resize and write with specified bitrate + + # Try different quality settings + for quality in [95, 80, 60, 40, 20]: + print(f"[FL_GeminiVideoCaptioner] Converting to WebM: {new_width}x{new_height}, quality={quality}") + try: - resized_video = video.resize(newsize=(new_width, new_height)) - resized_video.write_videofile( + # Create VideoWriter object + fourcc = cv2.VideoWriter_fourcc(*'VP80') # WebM codec + out = cv2.VideoWriter( output_path, - codec='libvpx', - audio=False, # No audio to keep file size down - bitrate=f"{bitrate}k", - logger=None # Suppress MoviePy progress bars + fourcc, + fps, + (new_width, new_height), + isColor=True ) - + + if not out.isOpened(): + raise Exception("Failed to open VideoWriter") + + frame_count = 0 + cap.set(cv2.CAP_PROP_POS_FRAMES, 0) # Reset to start + + while True: + ret, frame = cap.read() + if not ret or frame_count >= max_frames: + break + + # Resize frame if needed + if size_factor != 1.0: + frame = cv2.resize(frame, (new_width, new_height)) + + out.write(frame) + frame_count += 1 + + # Release VideoWriter + if out is not None: + out.release() + out = None + # Check file size file_size = os.path.getsize(output_path) if file_size <= max_size_bytes: - print(f"[FL_GeminiVideoCaptioner] Converted to WebM: {file_size/1024/1024:.2f}MB with dimensions {new_width}x{new_height}") - video.close() + print(f"[FL_GeminiVideoCaptioner] Created WebM: {file_size / 1024 / 1024:.2f}MB") return output_path - - print(f"[FL_GeminiVideoCaptioner] File too large ({file_size/1024/1024:.2f}MB), retrying with lower quality") + + print( + f"[FL_GeminiVideoCaptioner] File too large ({file_size / 1024 / 1024:.2f}MB), retrying with lower quality") + except Exception as e: print(f"[FL_GeminiVideoCaptioner] Error during conversion: {e}") + if out is not None: + out.release() + out = None continue - - # If we couldn't get under the size limit, return None - video.close() - if os.path.exists(output_path): - os.unlink(output_path) - print("[FL_GeminiVideoCaptioner] Couldn't compress video enough to meet size requirements") + + print("[FL_GeminiVideoCaptioner] Could not create WebM within size limit") return None - + except Exception as e: - print(f"[FL_GeminiVideoCaptioner] Error converting video: {e}") - if os.path.exists(output_path): - os.unlink(output_path) + print(f"[FL_GeminiVideoCaptioner] Error converting to WebM: {e}") return None - + + finally: + # Cleanup resources + if cap is not None: + cap.release() + if out is not None: + out.release() + if output_path and os.path.exists(output_path): + try: + os.unlink(output_path) + except: + pass + def create_webm_from_frames(self, frames, fps=30, max_size_mb=29): - """Create a WebM video from a list of frames using MoviePy - + """Create a WebM video from a list of frames using OpenCV + Args: frames: List of numpy arrays representing frames fps: Frames per second max_size_mb: Maximum size in MB for the output WebM file - + Returns: - Path to the created WebM file + Path to the created WebM file or None if creation fails """ if not frames: return None - - # Create a temporary file for the output WebM - with tempfile.NamedTemporaryFile(suffix=".webm", delete=False) as temp_webm: - output_path = temp_webm.name - + + output_path = None + out = None + try: + # Create a temporary file for the output WebM + with tempfile.NamedTemporaryFile(suffix=".webm", delete=False) as temp_webm: + output_path = temp_webm.name + # Get dimensions from the first frame height, width = frames[0].shape[:2] - - # Ensure dimensions are even (required for some codecs) + + # Ensure dimensions are even width = width if width % 2 == 0 else width + 1 height = height if height % 2 == 0 else height + 1 - + # Set size reduction factors and quality - size_factors = [1.0, 0.75, 0.5, 0.25, 0.125] # Progressive size reduction + size_factors = [1.0, 0.75, 0.5, 0.25, 0.125] max_size_bytes = max_size_mb * 1024 * 1024 - - # Make sure frames are in RGB format - rgb_frames = [] + + # Ensure frames are in correct format + processed_frames = [] for frame in frames: - # If frame is in BGR format (common with OpenCV), convert to RGB + if frame.dtype != np.uint8: + # Convert from float [0-1] to uint8 [0-255] if needed + frame = (frame * 255).astype(np.uint8) + # Convert BGR to RGB if needed if len(frame.shape) == 3 and frame.shape[2] == 3: - if frame.dtype != np.uint8: - # Convert from float [0-1] to uint8 [0-255] if needed - frame = (frame * 255).astype(np.uint8) - # Ensure the frame is in RGB format for MoviePy - if isinstance(frame, np.ndarray): - frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) if frame.dtype == np.uint8 else frame - rgb_frames.append(frame_rgb) - else: - rgb_frames.append(frame) - + frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) + processed_frames.append(frame) + + # Try different size factors and quality settings for size_factor in size_factors: # Calculate new dimensions new_width = int(width * size_factor) new_height = int(height * size_factor) - + # Ensure dimensions are even new_width = new_width if new_width % 2 == 0 else new_width + 1 new_height = new_height if new_height % 2 == 0 else new_height + 1 - - # Resize frames if needed - if size_factor != 1.0: - resized_frames = [] - for frame in rgb_frames: - resized = cv2.resize(frame, (new_width, new_height)) - resized_frames.append(resized) - else: - resized_frames = rgb_frames - - # Try different bitrates - for bitrate_factor in [1.0, 0.75, 0.5, 0.25, 0.1]: - # Estimate a reasonable bitrate based on resolution - base_bitrate = new_width * new_height * fps * 0.000005 - bitrate = int(base_bitrate * bitrate_factor) - - print(f"[FL_GeminiVideoCaptioner] Creating WebM: {new_width}x{new_height}, bitrate={bitrate}K") - + + # Try different quality settings + for quality in [95, 80, 60, 40, 20]: + print(f"[FL_GeminiVideoCaptioner] Creating WebM: {new_width}x{new_height}, quality={quality}") + try: - # Create clip from frames - clip = ImageSequenceClip(resized_frames, fps=fps) - - # Write to file - clip.write_videofile( + # Create VideoWriter object + fourcc = cv2.VideoWriter_fourcc(*'VP80') # WebM codec + out = cv2.VideoWriter( output_path, - codec='libvpx', - audio=False, - bitrate=f"{bitrate}k", - logger=None # Suppress MoviePy progress bars + fourcc, + fps, + (new_width, new_height), + isColor=True ) - + + if not out.isOpened(): + raise Exception("Failed to open VideoWriter") + + for frame in processed_frames: + # Resize frame if needed + if size_factor != 1.0: + frame = cv2.resize(frame, (new_width, new_height)) + + # Apply quality compression + encode_param = [int(cv2.IMWRITE_JPEG_QUALITY), quality] + _, encoded_frame = cv2.imencode('.jpg', frame, encode_param) + frame = cv2.imdecode(encoded_frame, cv2.IMREAD_COLOR) + + # Write frame + out.write(frame) + + # Release VideoWriter + if out is not None: + out.release() + out = None + # Check file size file_size = os.path.getsize(output_path) if file_size <= max_size_bytes: - print(f"[FL_GeminiVideoCaptioner] Created WebM: {file_size/1024/1024:.2f}MB with dimensions {new_width}x{new_height}") - clip.close() + print( + f"[FL_GeminiVideoCaptioner] Created WebM: {file_size / 1024 / 1024:.2f}MB with dimensions {new_width}x{new_height}") return output_path - - print(f"[FL_GeminiVideoCaptioner] File too large ({file_size/1024/1024:.2f}MB), retrying with lower quality") - clip.close() + + print( + f"[FL_GeminiVideoCaptioner] File too large ({file_size / 1024 / 1024:.2f}MB), retrying with lower quality") + except Exception as e: - print(f"[FL_GeminiVideoCaptioner] Error while creating clip: {e}") + print(f"[FL_GeminiVideoCaptioner] Error while creating video: {e}") + if out is not None: + out.release() + out = None continue - - # If we couldn't get under the size limit, return None + + # If we couldn't get under the size limit, cleanup and return None + print("[FL_GeminiVideoCaptioner] Couldn't create WebM within size limit") if os.path.exists(output_path): os.unlink(output_path) - print("[FL_GeminiVideoCaptioner] Couldn't create WebM within size limit") return None - + except Exception as e: print(f"[FL_GeminiVideoCaptioner] Error creating WebM from frames: {e}") - if os.path.exists(output_path): + if output_path and os.path.exists(output_path): os.unlink(output_path) return None - + + finally: + # Cleanup resources + if out is not None: + out.release() + def _send_api_request(self, api_url, payload): """Helper method to send API request and handle response""" try: @@ -684,12 +733,12 @@ class FL_GeminiVideoCaptioner: data=json.dumps(payload), timeout=300 # Longer timeout for video processing ) - + if response.status_code != 200: error_msg = f"API error: {response.status_code} - {response.text}" print(f"[FL_GeminiVideoCaptioner] {error_msg}") return f"Error: {error_msg}" - + result = response.json() if "candidates" in result and len(result["candidates"]) > 0: content = result["candidates"][0]["content"] @@ -701,9 +750,9 @@ class FL_GeminiVideoCaptioner: block_msg = f"Content blocked by Gemini API. Reason: {block_reason}" print(f"[FL_GeminiVideoCaptioner] {block_msg}") return f"Error: {block_msg}" - + return "Failed to get caption from Gemini API: unexpected response format" - + except requests.RequestException as e: error_msg = f"Network error during API call: {str(e)}" print(f"[FL_GeminiVideoCaptioner] {error_msg}")