diff --git a/README.md b/README.md index ed728db..c8a0eea 100644 --- a/README.md +++ b/README.md @@ -7,6 +7,10 @@ Through triples, any LLM model compatible with the OpenAI interface can be acces - **apikey** - **model** +## Update: Video reverse-prompt (video captioning) +- **Optional input `video`**: supports OpenAI-compatible `chat.completions` payload with `{"type":"video_url"}`. +- **No resize / no compression**: image keeps original resolution; video is read as-is and base64 encoded. + **Online Demo** https://www.runninghub.ai/post/1890402871119368194 ![image](https://github.com/user-attachments/assets/31b35db4-4d61-4767-a41c-6f1445fbea5e) diff --git a/node.py b/node.py index 8c99455..1603bc6 100644 --- a/node.py +++ b/node.py @@ -4,26 +4,83 @@ from PIL import Image import numpy as np import base64 import os +import io def encode_image_b64(ref_image): - i = 255. * ref_image.cpu().numpy()[0] + """ + Encode ComfyUI IMAGE tensor to base64 JPEG without resizing. + + Notes: + - Keep original resolution (no resize). + - Avoid temporary files (in-memory encoding). + """ + i = 255.0 * ref_image.cpu().numpy()[0] img = Image.fromarray(np.clip(i, 0, 255).astype(np.uint8)) - lsize = np.max(img.size) - factor = 1 - while lsize / factor > 2048: - factor *= 2 - img = img.resize((img.size[0] // factor, img.size[1] // factor)) + buf = io.BytesIO() + # Use JPEG to match the existing OpenAI-compatible payload mime label. + img.save(buf, format="JPEG", quality=95, optimize=True) + return base64.b64encode(buf.getvalue()).decode("utf-8") - image_path = f'{time.time()}.webp' - img.save(image_path, 'WEBP') - with open(image_path, "rb") as image_file: - base64_image = base64.b64encode(image_file.read()).decode('utf-8') +def _get_video_file_path(video): + """ + Try to extract a filesystem path from a ComfyUI VIDEO object. + Returns None if it cannot be resolved. + """ + # VideoFromFile type (private attribute) + if hasattr(video, "_VideoFromFile__file"): + path = getattr(video, "_VideoFromFile__file", None) + if isinstance(path, str) and os.path.exists(path): + return path - # print(img_base64) - os.remove(image_path) - return base64_image + # Stream-like sources + if hasattr(video, "get_stream_source"): + try: + stream_source = video.get_stream_source() + if isinstance(stream_source, str) and os.path.exists(stream_source): + return stream_source + except Exception: + pass + + # Common attributes + for attr in ("path", "file"): + if hasattr(video, attr): + path = getattr(video, attr, None) + if isinstance(path, str) and os.path.exists(path): + return path + + return None + + +def encode_video_b64(video): + """ + Encode ComfyUI VIDEO object to base64 MP4 bytes. + + Notes: + - No ffmpeg processing, no compression, no resizing. + - If a file path is available, read it directly. + - Otherwise, try saving via save_to() to a temp mp4 and read back. + """ + video_path = _get_video_file_path(video) + if video_path: + with open(video_path, "rb") as f: + return base64.b64encode(f.read()).decode("utf-8") + + if hasattr(video, "save_to"): + temp_path = f"temp_video_{time.time()}.mp4" + try: + video.save_to(temp_path) + with open(temp_path, "rb") as f: + return base64.b64encode(f.read()).decode("utf-8") + finally: + try: + if os.path.exists(temp_path): + os.remove(temp_path) + except Exception: + pass + + raise ValueError(f"Unable to read video data from object type: {type(video)}") class RH_LLMAPI_Node(): @@ -44,6 +101,7 @@ class RH_LLMAPI_Node(): }, "optional": { "ref_image": ("IMAGE",), + "video": ("VIDEO",), } } @@ -52,10 +110,30 @@ class RH_LLMAPI_Node(): FUNCTION = "rh_run_llmapi" CATEGORY = "Runninghub" - def rh_run_llmapi(self, api_baseurl, api_key, model, role, prompt, temperature, seed, ref_image=None): + def rh_run_llmapi(self, api_baseurl, api_key, model, role, prompt, temperature, seed, ref_image=None, video=None): client = OpenAI(api_key=api_key, base_url=api_baseurl) - if ref_image is None: + + # Priority: video > image > text (align with reference node behavior) + if video is not None: + base64_video = encode_video_b64(video) + messages = [ + {'role': 'system', 'content': f'{role}'}, + {'role': 'user', + 'content': [ + { + "type": "text", + "text": f"{prompt}" + }, + { + "type": "video_url", + "video_url": { + "url": f"data:video/mp4;base64,{base64_video}" + } + }, + ]}, + ] + elif ref_image is None: messages = [ {'role': 'system', 'content': f'{role}'}, {'role': 'user', 'content': f'{prompt}'},