From c8abe83cac4a965ed2e8723e4c76b1743cc74a82 Mon Sep 17 00:00:00 2001 From: MNeMoNiCuZ Date: Sun, 27 Sep 2026 20:52:56 +0200 Subject: [PATCH] Updated LLM Request node with improved UI, support for A Thousand Words VLM server, and other tweaks. --- .env.example | 6 + README.md | 2 +- README/{llm_api.md => llm_request.md} | 2 +- __init__.py | 2 +- nodes/__init__.py | 2 +- nodes/llm/DefaultEndpoints.json | 10 +- nodes/llm/UserEndpoints.example.json | 2 +- nodes/{llm_api.py => llm_request.py} | 20 +- pyproject.toml | 4 +- utils/llm_cli.py | 106 ++++++-- utils/llm_endpoints.py | 2 +- utils/llm_providers.py | 106 +++++++- utils/llm_routes.py | 2 +- utils/settings_utils.py | 5 +- web/docs/MNeMiC_GroqAPILLM.md | 4 +- web/docs/MNeMiC_LLMAPI.md | 44 ++-- web/js/{llm_api.js => llm_request.js} | 354 +++++++++++++++++--------- web/js/settings.js | 17 +- 18 files changed, 495 insertions(+), 195 deletions(-) rename README/{llm_api.md => llm_request.md} (99%) rename nodes/{llm_api.py => llm_request.py} (93%) rename web/js/{llm_api.js => llm_request.js} (77%) diff --git a/.env.example b/.env.example index ba9e99f..2af1dff 100644 --- a/.env.example +++ b/.env.example @@ -50,6 +50,12 @@ OPENAI_COMPATIBLE_API_KEY= CLAUDE_CLI= CODEX_CLI= +# --- A Thousand Words --------------------------------------------------------- +# Local captioning server (server.bat). Only needed if it is not on the +# default port. +# default http://127.0.0.1:8585 +ATHOUSANDWORDS_URL= + # --- Sanctum ------------------------------------------------------------------- # Server address (http://host:port) SANCTUM_URL= diff --git a/README.md b/README.md index 86f888e..26e8b58 100644 --- a/README.md +++ b/README.md @@ -141,7 +141,7 @@ Visual bounding-box for Ideogram 4 with added automatic string input for each re Automatically generates random Ideogram 4 json-structured compositions with dictionary-sourced descriptions. image -## ✨🧠 [LLM Request](./README/llm_api.md) +## ✨ [LLM Request](./README/llm_request.md) One node for every LLM: ChatGPT, Claude, Gemini, Grok, Groq, OpenRouter, Mistral, DeepSeek, Ollama / LM Studio / any OpenAI-compatible server on your PC or your network, and your Claude Code or Codex subscription. Model browser, live streaming preview, vision, reasoning control, and live config support. diff --git a/README/llm_api.md b/README/llm_request.md similarity index 99% rename from README/llm_api.md rename to README/llm_request.md index ddb9d5c..4eb6e29 100644 --- a/README/llm_api.md +++ b/README/llm_request.md @@ -1,4 +1,4 @@ -# ✨🧠 LLM Request +# ✨ LLM Request One node for every language model: ChatGPT, Claude, Gemini, Grok, Groq, OpenRouter, Mistral, DeepSeek, local servers like Ollama and LM Studio (on diff --git a/__init__.py b/__init__.py index b4f22eb..b080803 100644 --- a/__init__.py +++ b/__init__.py @@ -10,7 +10,7 @@ from .nodes.groq_api_llm import GroqAPILLM from .nodes.groq_api_vlm import GroqAPIVLM from .nodes.groq_api_alm_transcribe import GroqAPIALMTranscribe #from .nodes.groq_api_alm_translate import GroqAPIALMTranslate -from .nodes.llm_api import LLMAPI +from .nodes.llm_request import LLMAPI from .nodes.tiktoken_tokenizer import TiktokenTokenizer from .nodes.string_cleaning import StringCleaning from .nodes.generate_negative_prompt import GenerateNegativePrompt diff --git a/nodes/__init__.py b/nodes/__init__.py index 7042133..18b7b32 100644 --- a/nodes/__init__.py +++ b/nodes/__init__.py @@ -5,7 +5,7 @@ from .get_file_path import GetFilePath from .groq_api_llm import GroqAPILLM from .groq_api_vlm import GroqAPIVLM from .groq_api_alm_transcribe import GroqAPIALMTranscribe -from .llm_api import LLMAPI +from .llm_request import LLMAPI from .tiktoken_tokenizer import TiktokenTokenizer from .string_cleaning import StringCleaning from .lora_tag_loader import LoraTagLoader diff --git a/nodes/llm/DefaultEndpoints.json b/nodes/llm/DefaultEndpoints.json index 50078fe..dcbc094 100644 --- a/nodes/llm/DefaultEndpoints.json +++ b/nodes/llm/DefaultEndpoints.json @@ -30,7 +30,7 @@ "description": "Any OpenAI-compatible server: llama.cpp, vLLM, KoboldCpp, text-generation-webui, LocalAI, Jan… Set OPENAI_COMPATIBLE_URL in .env (usually ending in /v1)." }, { - "name": "Local Claude Code Subscription", + "name": "Claude Code Subscription", "provider": "claude_cli", "command": "${CLAUDE_CLI:-claude}", "options": { @@ -39,7 +39,7 @@ "description": "Runs the Claude Code CLI on this PC (claude -p) with your Claude subscription. Install Claude Code and sign in once by running `claude` in a terminal. Empty model uses Claude Code's default." }, { - "name": "Local Codex Subscription", + "name": "Codex Subscription", "provider": "codex_cli", "command": "${CODEX_CLI:-codex}", "options": { @@ -117,6 +117,12 @@ "default_model": "deepseek-chat", "description": "DeepSeek's API. Key from platform.deepseek.com." }, + { + "name": "A Thousand Words", + "provider": "athousandwords", + "base_url": "${ATHOUSANDWORDS_URL:-http://127.0.0.1:8585}", + "description": "A Thousand Words running on this PC (server.bat). A dedicated image/video captioning server: requires images or a video. No key needed." + }, { "name": "Sanctum", "provider": "openai", diff --git a/nodes/llm/UserEndpoints.example.json b/nodes/llm/UserEndpoints.example.json index 0791589..d3cf572 100644 --- a/nodes/llm/UserEndpoints.example.json +++ b/nodes/llm/UserEndpoints.example.json @@ -1,5 +1,5 @@ { - "_readme": "Your own endpoints for the ✨🧠 LLM Request node. Copied to UserEndpoints.json on first run; that copy is git-ignored and survives updates. Entries here are added to the built-in list; one with the same name as a built-in replaces it, and \"enabled\": false hides it. Never paste keys or private addresses here: put them in the .env file in the pack root and reference them as ${VAR}. Press R in ComfyUI (refresh node definitions) after editing.", + "_readme": "Your own endpoints for the ✨ LLM Request node. Copied to UserEndpoints.json on first run; that copy is git-ignored and survives updates. Entries here are added to the built-in list; one with the same name as a built-in replaces it, and \"enabled\": false hides it. Never paste keys or private addresses here: put them in the .env file in the pack root and reference them as ${VAR}. Press R in ComfyUI (refresh node definitions) after editing.", "endpoints": [ { "name": "My llama.cpp box", diff --git a/nodes/llm_api.py b/nodes/llm_request.py similarity index 93% rename from nodes/llm_api.py rename to nodes/llm_request.py index 3a180ed..5a8f99f 100644 --- a/nodes/llm_api.py +++ b/nodes/llm_request.py @@ -101,12 +101,12 @@ class LLMAPI(io.ComfyNode): return io.Schema( node_id="MNeMiC_LLMAPI", - display_name="✨🧠 LLM Request", + display_name="✨ LLM Request", category="⚡ MNeMiC Nodes", - description="Sends a prompt, and optionally images, to any LLM: ChatGPT, Claude, Gemini, Grok, Groq, OpenRouter, or Ollama and LM Studio on this PC or your network.", + description="Sends a prompt, and optionally images or a video, to any LLM: ChatGPT, Claude, Gemini, Grok, Groq, OpenRouter, A Thousand Words, or Ollama and LM Studio on this PC or your network.", search_aliases=["llm", "chat", "ollama", "openai", "chatgpt", "gpt", "claude", "anthropic", "gemini", "grok", "xai", "groq", "openrouter", "lm studio", "llama.cpp", "vllm", "vlm", "prompt generator", - "universal llm api"], + "universal llm api", "a thousand words", "caption", "video caption"], inputs=[ io.Combo.Input("endpoint", options=names, default=names[0], tooltip="Which server to talk to. Endpoints are defined in nodes/llm/*.json; keys and private addresses come from .env and are never saved in the workflow."), @@ -120,6 +120,8 @@ class LLMAPI(io.ComfyNode): tooltip="The request itself: what you want the model to write, rewrite or describe. May be empty: the system message or preset is then sent on its own."), io.Image.Input("images", optional=True, tooltip="Images to send along with the prompt, for vision models. Every image in the batch is sent."), + io.Video.Input("video", optional=True, + tooltip="A Thousand Words only: a video to caption instead of, or alongside, images. Which models accept video depends on the server's own model list."), io.Float.Input("temperature", default=0.8, min=0.0, max=2.0, step=0.05, tooltip="Randomness. Low is focused and repeatable, high is varied and creative. Dropped automatically for models that only allow their default."), io.Combo.Input("reasoning", options=REASONING_LEVELS, default="default", advanced=True, @@ -167,7 +169,7 @@ class LLMAPI(io.ComfyNode): async def execute(cls, endpoint, model, preset, system_message, user_input, temperature, reasoning="default", max_tokens=0, top_p=1.0, seed=42, stop="", json_mode=False, unload_model_after=False, context_length=0, free_comfy_vram=False, max_retries=2, - raise_on_error=True, custom_endpoint="", images=None) -> io.NodeOutput: + raise_on_error=True, custom_endpoint="", images=None, video=None) -> io.NodeOutput: node_id = cls.hidden.unique_id if cls.hidden else None client_id = _current_client_id() console_log = is_llm_console_log_enabled() @@ -180,7 +182,7 @@ class LLMAPI(io.ComfyNode): if raise_on_error: # from None: the chained original error would otherwise be # printed in ComfyUI's traceback. - raise RuntimeError(f"✨🧠 LLM Request — {message}") from None + raise RuntimeError(f"✨ LLM Request — {message}") from None return io.NodeOutput("", "", False, message, ui={"mnemic_llm": [{"ok": False, "error": message, "status": status}]}) @@ -196,6 +198,8 @@ class LLMAPI(io.ComfyNode): ep = ep_config.resolve() if not ep.ok: return fail(f"{endpoint}: {' '.join(ep.problems)}", "not configured") + if video is not None and ep_config.provider != "athousandwords": + return fail(f"{endpoint} can't take a video; only A Thousand Words can.") model = (model or "").strip() or ep_config.default_model # Local and network servers (Ollama, LM Studio, llama.cpp…) usually @@ -231,7 +235,10 @@ class LLMAPI(io.ComfyNode): caption_block = "\n\n".join(f"[Image {i + 1} description]: {d}" for i, d in enumerate(descriptions) if d) user_input = f"{caption_block}\n\n{user_input}".strip() if user_input else caption_block pil_images = [] - if not user_input and not pil_images: + has_media = bool(pil_images) or video is not None + if ep_config.provider == "athousandwords" and not has_media: + return fail("A Thousand Words requires at least one image or a video.") + if not user_input and not has_media: if not system_message: return fail("There is nothing to send: system_message, user_input and images are all empty.") # Instructions alone are a valid request; most APIs need a user @@ -243,6 +250,7 @@ class LLMAPI(io.ComfyNode): system=system_message, user=user_input, images=pil_images, + videos=[video] if video is not None else [], temperature=temperature, top_p=top_p, max_tokens=max_tokens, diff --git a/pyproject.toml b/pyproject.toml index 227e947..9850644 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "comfyui-mnemic-nodes" -description = "Added LLM Request node and made the wildcard processor node show a preview, and colorize special inputs" -version = "3.0.3" +description = "Updated LLM Request node with improved UI, support for A Thousand Words VLM server, and other tweaks." +version = "3.0.4" license = { file = "LICENSE" } dependencies = ["configparser", "groq", "transformers", "torch", "tiktoken", "imageio", "tqdm", "piexif", "requests", "colorama", "opencv-python", "python-dotenv"] diff --git a/utils/llm_cli.py b/utils/llm_cli.py index 2257f03..36786cd 100644 --- a/utils/llm_cli.py +++ b/utils/llm_cli.py @@ -397,27 +397,99 @@ def run_cli_chat(provider, command, req, result, **kwargs): raise CLIError(f"could not start the CLI ({type(e).__name__})", status="cli error") from None +# Every model here is current-generation and accepts image input, so all are +# marked vision: True. static: True marks a fixed list, not fetched live, for +# the picker's "*" indicator. Only the rolling alias is listed for each model +# (not also its current pinned full name, e.g. claude-sonnet-5): both resolve +# to the same model today, and the alias is the one that keeps working as +# Anthropic ships new versions. CLAUDE_MODELS = [ - {"id": "sonnet", "detail": "alias for claude-sonnet-5"}, - {"id": "claude-sonnet-5", "detail": "Sonnet 5"}, - {"id": "opus", "detail": "alias for claude-opus-5-5"}, - {"id": "claude-opus-5-5", "detail": "Opus 5.5"}, - {"id": "haiku", "detail": "alias for claude-haiku-4-5-20251001"}, - {"id": "claude-haiku-4-5-20251001", "detail": "Haiku 4.5"}, - {"id": "fable", "detail": "alias for claude-fable-5-1"}, - {"id": "claude-fable-5-1", "detail": "Fable 5.1"}, + {"id": "sonnet", "detail": "", "vision": True, "static": True}, + {"id": "opus", "detail": "", "vision": True, "static": True}, + {"id": "haiku", "detail": "", "vision": True, "static": True}, + {"id": "fable", "detail": "", "vision": True, "static": True}, ] # Codex has no equivalent of `claude` picking up new releases under a fixed -# alias, and no command to list what a given install supports (openai/codex#8871 -# asks for exactly this); these are current model names, not a live list. +# alias. Its app-server does expose a live model/list RPC (see +# _codex_model_list_rpc below), so this is only the fallback when that can't +# be reached (CLI not installed, too old to speak the protocol, timed out…). +# Matched against codex-cli 0.157.0's own live catalog, not a guess. CODEX_MODELS = [ - {"id": "gpt-5.2-codex", "detail": "current Codex model"}, - {"id": "gpt-5.1-codex-max", "detail": "previous Codex model"}, - {"id": "gpt-5.1-codex-mini", "detail": "smaller, faster"}, + {"id": "gpt-6-astra", "detail": "", "vision": True, "static": True}, + {"id": "gpt-6-sol", "detail": "", "vision": True, "static": True}, + {"id": "gpt-6-luna", "detail": "", "vision": True, "static": True}, + {"id": "gpt-5.6-sol", "detail": "", "vision": True, "static": True}, + {"id": "gpt-5.6-terra", "detail": "", "vision": True, "static": True}, + {"id": "gpt-5.6-luna", "detail": "", "vision": True, "static": True}, + {"id": "gpt-5.5", "detail": "", "vision": True, "static": True}, ] -def list_cli_models(provider): - """Neither CLI can list what a given install actually supports; these are - known model names/aliases, not fetched live.""" - return list(CLAUDE_MODELS) if provider == CLAUDE else list(CODEX_MODELS) +def _codex_model_list_rpc(command, timeout=6): + """The live model catalog from Codex's own `app-server` JSON-RPC daemon + (method "model/list"), so the picker matches what this install actually + offers. None on any failure (not installed, too old to speak this + protocol, no reply in time…). + """ + try: + with _work_dir() as cwd: + proc = _popen([command, "app-server"], CODEX, cwd) + lines = queue.Queue() + threading.Thread(target=_read_lines, args=(proc.stdout, lines), daemon=True).start() + try: + proc.stdin.write((json.dumps({ + "jsonrpc": "2.0", "id": 1, "method": "initialize", + "params": {"clientInfo": {"name": "ComfyUI-mnemic-nodes", "version": "1.0"}}, + }) + "\n").encode()) + proc.stdin.write((json.dumps({ + "jsonrpc": "2.0", "id": 2, "method": "model/list", "params": {}, + }) + "\n").encode()) + proc.stdin.flush() + + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + try: + raw = lines.get(timeout=0.25) + except queue.Empty: + continue + if raw is None: + return None + try: + msg = json.loads(raw.decode("utf-8", errors="replace")) + except ValueError: + continue + if msg.get("id") == 2: + return (msg.get("result") or {}).get("data") + return None + finally: + _terminate(proc) + except OSError: + return None + + +def _codex_models_from_catalog(data): + models = [] + for m in data or []: + if m.get("hidden"): + continue + model_id = m.get("id") or m.get("model") + if not model_id: + continue + vision = "image" in (m.get("inputModalities") or ["text", "image"]) + models.append({"id": model_id, "detail": "", "vision": vision}) + return models + + +def list_cli_models(provider, command=""): + """Claude Code has no documented way to enumerate installed models, so + CLAUDE_MODELS is a fixed list of current aliases/names. Codex exposes a + live catalog through its own app-server; that is queried directly, and + CODEX_MODELS is only the fallback when it can't be reached. + """ + if provider == CLAUDE: + return list(CLAUDE_MODELS) + if command: + models = _codex_models_from_catalog(_codex_model_list_rpc(command)) + if models: + return models + return list(CODEX_MODELS) diff --git a/utils/llm_endpoints.py b/utils/llm_endpoints.py index cc2c1ac..eb842bf 100644 --- a/utils/llm_endpoints.py +++ b/utils/llm_endpoints.py @@ -25,7 +25,7 @@ DEFAULT_ENDPOINTS_FILE = os.path.join(LLM_DIR, "DefaultEndpoints.json") USER_ENDPOINTS_FILE = os.path.join(LLM_DIR, "UserEndpoints.json") USER_ENDPOINTS_EXAMPLE = os.path.join(LLM_DIR, "UserEndpoints.example.json") -PROVIDERS = ("openai", "anthropic", "ollama", "claude_cli", "codex_cli") +PROVIDERS = ("openai", "anthropic", "ollama", "claude_cli", "codex_cli", "athousandwords") CLI_PROVIDERS = ("claude_cli", "codex_cli") CUSTOM_ENDPOINT_NAME = "Custom Endpoint - WARNING" diff --git a/utils/llm_providers.py b/utils/llm_providers.py index da23b8e..a17e410 100644 --- a/utils/llm_providers.py +++ b/utils/llm_providers.py @@ -12,6 +12,10 @@ Each adapter turns a ChatRequest into (url, headers, body), and turns the reply — whole or streamed — back into text. `run_chat` does the HTTP around it: retries, backoff, live streaming, interrupts, and dropping parameters an endpoint rejects. + +`athousandwords` is not a chat API (one multipart POST /caption per request, +no streaming, images/video only) and bypasses the adapter's build/parse: see +_run_athousandwords. Its adapter only serves list_models. """ import base64 @@ -61,6 +65,7 @@ class ChatRequest: system: str = "" user: str = "" images: list = field(default_factory=list) # PIL images + videos: list = field(default_factory=list) # VIDEO inputs; A Thousand Words only temperature: float | None = None top_p: float | None = None max_tokens: int = 0 @@ -90,8 +95,8 @@ class ChatResult: # Helpers # -------------------------------------------------------------------------- -def encode_images(images): - """PIL images → list of (mime, base64). Large images are scaled down.""" +def _encode_image_bytes(images): + """PIL images → list of (mime, raw bytes). Large images are scaled down.""" encoded = [] for image in images: image = image.convert("RGB") @@ -99,10 +104,28 @@ def encode_images(images): image.thumbnail((MAX_IMAGE_SIDE, MAX_IMAGE_SIDE)) buffer = BytesIO() image.save(buffer, format="JPEG", quality=92) - encoded.append(("image/jpeg", base64.b64encode(buffer.getvalue()).decode("ascii"))) + encoded.append(("image/jpeg", buffer.getvalue())) return encoded +def encode_images(images): + """PIL images → list of (mime, base64). Large images are scaled down.""" + return [(mime, base64.b64encode(data).decode("ascii")) for mime, data in _encode_image_bytes(images)] + + +_VIDEO_MIME = {"mp4": "video/mp4", "avi": "video/x-msvideo", "mov": "video/quicktime", + "matroska": "video/x-matroska", "webm": "video/webm"} + + +def encode_video(video): + """A ComfyUI VIDEO input → (filename, raw bytes, mime), without re-encoding.""" + fmt = (video.get_container_format() or "mp4").lower() + ext, mime = next(((e, m) for e, m in _VIDEO_MIME.items() if e in fmt), ("mp4", "video/mp4")) + source = video.get_stream_source() + data = open(source, "rb").read() if isinstance(source, str) else source.getvalue() + return f"video.{ext}", data, mime + + _THINK_BLOCK = re.compile(r"<(think|thinking|reasoning)>(.*?)", re.DOTALL | re.IGNORECASE) _THINK_OPEN_TAG = re.compile(r"<(think|thinking|reasoning)>", re.IGNORECASE) _THINK_OPEN_ONLY = re.compile(r"^\s*<(think|thinking|reasoning)>(.*)$", re.DOTALL | re.IGNORECASE) @@ -470,7 +493,10 @@ class AnthropicAdapter: def list_models(self, ep, timeout): response = requests.get(f"{ep.base_url}/models", params={"limit": 1000}, headers=self.headers(ep), timeout=timeout, allow_redirects=False) _raise_for_status(response) - return [{"id": m["id"], "detail": m.get("display_name", "")} for m in response.json().get("data", [])] + # Every model the Anthropic API lists is a current Claude model, and + # all of them accept image input. + return [{"id": m["id"], "detail": m.get("display_name", ""), "vision": True} + for m in response.json().get("data", [])] class OllamaAdapter: @@ -565,11 +591,38 @@ class OllamaAdapter: detail = [d for d in (details.get("parameter_size"), details.get("quantization_level")) if d] if item.get("size"): detail.append(f"{item['size'] / 1e9:.1f} GB") - models.append({"id": item["name"], "detail": " · ".join(detail), "loaded": item["name"] in loaded}) + vision = "vision" in (item.get("capabilities") or []) + models.append({"id": item["name"], "detail": " · ".join(detail), + "loaded": item["name"] in loaded, "vision": vision}) return sorted(models, key=lambda m: (not m["loaded"], m["id"].lower())) -ADAPTERS = {a.name: a for a in (OpenAIAdapter(), AnthropicAdapter(), OllamaAdapter())} +class AThousandWordsAdapter: + """A Thousand Words is a captioning server, not a chat API: one POST + /caption per request, multipart/form-data, no streaming. run_chat sends + it through _run_athousandwords instead of this adapter's build/parse; it + only serves list_models here.""" + name = "athousandwords" + + def headers(self, ep): + return dict(ep.headers) + + def list_models(self, ep, timeout): + response = requests.get(f"{ep.base_url}/models", headers=self.headers(ep), timeout=timeout, allow_redirects=False) + _raise_for_status(response) + data = response.json() + batch_sizes = data.get("batch_sizes") or {} + models = [] + for model_id in data.get("models", []): + recommended = (batch_sizes.get(model_id) or {}).get("recommended") + # The server doesn't say which models take video; every model here + # accepts the same /caption call, images or video, so both are marked. + models.append({"id": model_id, "detail": f"batch {recommended}" if recommended else "", + "vision": True, "video": True}) + return sorted(models, key=lambda m: m["id"].lower()) + + +ADAPTERS = {a.name: a for a in (OpenAIAdapter(), AnthropicAdapter(), OllamaAdapter(), AThousandWordsAdapter())} # CLI providers (claude_cli, codex_cli) have no HTTP adapter: see llm_cli. @@ -692,6 +745,8 @@ def run_chat(ep, req, *, timeout=300, max_retries=2, on_delta=None, check_interr """ if ep.endpoint.is_cli: return _run_cli(ep, req, timeout, on_delta, check_interrupt, log) + if ep.endpoint.provider == "athousandwords": + return _run_athousandwords(ep, req, timeout, check_interrupt, log) adapter = get_adapter(ep.endpoint.provider) url = adapter.chat_url(ep) headers = adapter.headers(ep) @@ -858,11 +913,48 @@ def _run_cli(ep, req, timeout, on_delta, check_interrupt, log): return result +def _run_athousandwords(ep, req, timeout, check_interrupt, log): + """POST /caption: multipart/form-data, one reply per call, no streaming.""" + result = ChatResult() + started = time.monotonic() + files = [("files", (f"image{i}.jpg", data, mime)) + for i, (mime, data) in enumerate(_encode_image_bytes(req.images))] + files += [("files", encode_video(video)) for video in req.videos] + + form = {"model": req.model} + task_prompt = "\n\n".join(t for t in (req.system, req.user) if t) + if task_prompt: + form["task_prompt"] = task_prompt + if req.max_tokens: + form["max_tokens"] = str(req.max_tokens) + if req.temperature is not None: + form["temperature"] = str(req.temperature) + + url = f"{ep.base_url}/caption" + if check_interrupt: + check_interrupt() + if log: + log(f"POST {url}\n{json.dumps({**form, 'files': f'{len(files)} file(s)'}, indent=2)}") + try: + response = requests.post(url, headers=ep.headers, data=form, files=files, + allow_redirects=False, timeout=(15, timeout)) + except (requests.RequestException, LocationValueError) as e: + raise LLMError(f"Could not reach {ep.endpoint.name}: {_short_exception(e)}", status="connection error") + _raise_for_status(response) + try: + data = response.json() + result.text = "\n\n".join(r.get("caption", "") for r in data.get("results", [])) + except (ValueError, AttributeError) as e: + raise LLMError(f"Bad response from {ep.endpoint.name}: {_short_exception(e)}", status="bad response") + result.seconds = time.monotonic() - started + return result + + def list_models(ep, timeout=15): """Models the endpoint reports, falling back to the configured list.""" if ep.endpoint.is_cli: from .llm_cli import list_cli_models - return list_cli_models(ep.endpoint.provider) + return list_cli_models(ep.endpoint.provider, ep.base_url) adapter = get_adapter(ep.endpoint.provider) try: return adapter.list_models(ep, timeout) diff --git a/utils/llm_routes.py b/utils/llm_routes.py index 128c115..8f690cd 100644 --- a/utils/llm_routes.py +++ b/utils/llm_routes.py @@ -1,4 +1,4 @@ -"""HTTP routes behind the Universal LLM node's UI (web/js/llm_api.js). +"""HTTP routes behind the Universal LLM node's UI (web/js/llm_request.js). Endpoints are addressed by name only; the browser never receives a key or a header value. Model listing happens here, server-side, because the key must diff --git a/utils/settings_utils.py b/utils/settings_utils.py index 65978ec..800a3b9 100644 --- a/utils/settings_utils.py +++ b/utils/settings_utils.py @@ -149,8 +149,8 @@ LLM_ENDPOINT_VISIBILITY_IDS = { "Ollama (network)": "MNeMiC.LLM.ShowEndpoint.OllamaNetwork", "LM Studio (this PC)": "MNeMiC.LLM.ShowEndpoint.LMStudio", "OpenAI-compatible server": "MNeMiC.LLM.ShowEndpoint.OpenAICompatible", - "Local Claude Code Subscription": "MNeMiC.LLM.ShowEndpoint.ClaudeCodeSubscription", - "Local Codex Subscription": "MNeMiC.LLM.ShowEndpoint.CodexSubscription", + "Claude Code Subscription": "MNeMiC.LLM.ShowEndpoint.ClaudeCodeSubscription", + "Codex Subscription": "MNeMiC.LLM.ShowEndpoint.CodexSubscription", "OpenAI (ChatGPT)": "MNeMiC.LLM.ShowEndpoint.OpenAI", "Anthropic (Claude)": "MNeMiC.LLM.ShowEndpoint.Anthropic", "Google (Gemini)": "MNeMiC.LLM.ShowEndpoint.Gemini", @@ -159,6 +159,7 @@ LLM_ENDPOINT_VISIBILITY_IDS = { "OpenRouter": "MNeMiC.LLM.ShowEndpoint.OpenRouter", "Mistral": "MNeMiC.LLM.ShowEndpoint.Mistral", "DeepSeek": "MNeMiC.LLM.ShowEndpoint.DeepSeek", + "A Thousand Words": "MNeMiC.LLM.ShowEndpoint.AThousandWords", "Sanctum": "MNeMiC.LLM.ShowEndpoint.Sanctum", "Custom Endpoint - WARNING": "MNeMiC.LLM.ShowEndpoint.CustomEndpoint", } diff --git a/web/docs/MNeMiC_GroqAPILLM.md b/web/docs/MNeMiC_GroqAPILLM.md index dfb25aa..dc67c65 100644 --- a/web/docs/MNeMiC_GroqAPILLM.md +++ b/web/docs/MNeMiC_GroqAPILLM.md @@ -15,7 +15,7 @@ GROQ_API_KEY=your_key_here ``` Changes apply on the next run; no restart needed. The same key is used by the -✨🧠 LLM Request node's Groq endpoint. +✨ LLM Request node's Groq endpoint. ## Inputs @@ -49,7 +49,7 @@ Advanced: Presets come from `nodes/groq/DefaultPrompts.json` (shipped) and `nodes/groq/UserPrompts.json` (yours). Each entry is `{"name": ..., "content": -...}`; the name shows in the dropdown. They are shared with the ✨🧠 Universal +...}`; the name shows in the dropdown. They are shared with the ✨ Universal LLM API node. Press R in ComfyUI (refresh node definitions) after editing. ## Notes diff --git a/web/docs/MNeMiC_LLMAPI.md b/web/docs/MNeMiC_LLMAPI.md index 775782a..ed7b946 100644 --- a/web/docs/MNeMiC_LLMAPI.md +++ b/web/docs/MNeMiC_LLMAPI.md @@ -1,10 +1,10 @@ -# ✨🧠 LLM Request +# ✨ LLM Request -Sends a prompt, and optionally images, to any language model and returns the -reply. One node covers cloud APIs (ChatGPT, Claude, Gemini, Grok, Groq, -OpenRouter, Mistral, DeepSeek) and local servers (Ollama and LM Studio on this -PC, Ollama or any OpenAI-compatible server on your network). Presets are -shared with the Groq nodes. +Sends a prompt, and optionally images or a video, to any language model and +returns the reply. One node covers cloud APIs (ChatGPT, Claude, Gemini, Grok, +Groq, OpenRouter, Mistral, DeepSeek) and local servers (Ollama and LM Studio on +this PC, Ollama or any OpenAI-compatible server on your network, A Thousand +Words for image/video captioning). Presets are shared with the Groq nodes. ## Setup @@ -27,22 +27,27 @@ and what is missing if not. ## Inputs -- **endpoint** — Which server to call. The list comes from - `nodes/llm/DefaultEndpoints.json` and your `nodes/llm/UserEndpoints.json`. -- **model** — Model name. Empty uses the endpoint's default model. Endpoints - on this PC or your network without one use the first chat model the - server lists, skipping embedding models (for Ollama, a model already in - memory if there is one, else the first installed one alphabetically); - cloud endpoints without one need a model chosen. Click **🔍 Models** to browse and search what the - endpoint offers; Ollama models already in memory are marked. -- **preset** — A saved system prompt, or the first entry to use - `system_message`. Click **📜 Preset** to read the selected one. -- **system_message** — The model's instructions. Ignored while a preset is - active; in the classic node view it is also greyed out and shows the - preset's text as a hint (📜 Preset shows it in either view). +- **endpoint** — Which server to call. Click it (or the ▾) to browse and + search the list, in the order `nodes/llm/DefaultEndpoints.json` and your + `nodes/llm/UserEndpoints.json` define them. +- **model** — Model name. Type one directly, or click the ▾ to browse and + search what the endpoint offers (Ollama models already in memory are + marked). Empty uses the endpoint's default model. Endpoints on this PC or + your network without one use the first chat model the server lists, + skipping embedding models (for Ollama, a model already in memory if there + is one, else the first installed one alphabetically); cloud endpoints + without one need a model chosen. +- **preset** — A saved system prompt. Picking one copies its text into + `system_message` (asking first if that would overwrite something different) + and resets itself back to the first entry, so `system_message` stays a + plain, freely editable field afterward. - **user_input** — The request. May be empty: the system message (or preset) is then sent on its own as the request. - **images** — Optional. Every image in the batch is sent, for vision models. +- **video** — Optional, A Thousand Words only. A single video to caption + instead of, or alongside, images. Sending it to any other endpoint fails: + no other protocol here accepts video. Which of the server's models actually + support video input is up to the server; it isn't in its model list. - **temperature** — Randomness. Dropped automatically for models that refuse anything but their default. @@ -96,6 +101,7 @@ back in an error message is masked before it is shown or returned. | `ollama` | Ollama's native API (for `keep_alive`, `num_ctx`, `think`) | | `claude_cli` | The Claude Code CLI on this PC, with your Claude subscription | | `codex_cli` | The Codex CLI on this PC, with your ChatGPT subscription | +| `athousandwords` | A Thousand Words, a local image/video captioning server (not a chat API: one `POST /caption` per call, `system_message` and `user_input` are combined into its `task_prompt`, no streaming) | **Parameters that don't fit.** Models differ in what they accept: OpenAI's reasoning models refuse `temperature`, some servers don't know `seed`. For diff --git a/web/js/llm_api.js b/web/js/llm_request.js similarity index 77% rename from web/js/llm_api.js rename to web/js/llm_request.js index 9c8b85b..af6862c 100644 --- a/web/js/llm_api.js +++ b/web/js/llm_request.js @@ -1,7 +1,7 @@ import { app } from "../../../scripts/app.js"; import { api } from "../../../scripts/api.js"; -// ✨🧠 LLM Request — node UI. +// ✨ LLM Request — node UI. // // Adds a panel to the node with the endpoint's status (where it runs, whether // its key/address is set), a searchable model browser, a connection test, a @@ -28,10 +28,10 @@ function defaultOutHeight() { } const LOCATION_LABEL = { - local: ["🖥", "This PC"], - network: ["🏠", "Network"], - cloud: ["☁", "Cloud"], - unknown: ["❔", "Not set"], + local: "This PC", + network: "Network", + cloud: "Cloud", + unknown: "Not configured", }; const panels = new Set(); @@ -97,32 +97,6 @@ function splitInlineThinking(text) { return [text.slice(m[0].length).trimStart(), m[2].trim()]; } -// navigator.clipboard only exists in secure contexts; ComfyUI opened over -// http:// is not one, so fall back to the old selection copy. -async function copyText(text) { - try { - if (navigator.clipboard?.writeText) { - await navigator.clipboard.writeText(text); - return true; - } - } catch { - // fall through to the fallback - } - const area = document.createElement("textarea"); - area.value = text; - area.style.cssText = "position:fixed;left:-9999px;top:0;opacity:0"; - document.body.appendChild(area); - area.select(); - let ok = false; - try { - ok = document.execCommand("copy"); - } catch { - ok = false; - } - area.remove(); - return ok; -} - // -------------------------------------------------------------------------- // Styles // -------------------------------------------------------------------------- @@ -142,7 +116,20 @@ function addStylesheet() { .mnemic-llm-chip { padding:1px 7px; border-radius:10px; background:var(--comfy-input-bg); border:1px solid var(--border-color); white-space:nowrap; font-size:11px; } .mnemic-llm-chip.bad { border-color:#d29922; color:#d29922; } - .mnemic-llm-host { opacity:.6; font-size:11px; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; min-width:0; flex:1; } + .mnemic-llm-host { opacity:.6; font-size:11px; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; min-width:0; } + .mnemic-llm-test { flex:0 0 auto; min-width:0; margin-left:auto; padding:2px 8px; } + .mnemic-llm-pickrow { display:flex; gap:6px; } + .mnemic-llm-pick { flex:1; min-width:0; display:flex; align-items:center; gap:4px; padding:3px 6px; border-radius:5px; + background:var(--comfy-input-bg); color:var(--fg-color); border:1px solid var(--border-color); cursor:pointer; } + .mnemic-llm-pick:hover { border-color:#58a6ff; } + .mnemic-llm-pick.mnemic-llm-pick-combo { padding:0; cursor:default; } + .mnemic-llm-pick-value { flex:1; min-width:0; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; text-align:left; font-size:11.5px; } + .mnemic-llm-pick-input { flex:1; min-width:0; padding:3px 0 3px 6px; border:none; background:none; color:var(--fg-color); font-size:11.5px; } + .mnemic-llm-pick-input:focus { outline:none; } + .mnemic-llm-arrow { opacity:.6; font-size:10px; flex:0 0 auto; } + .mnemic-llm-arrow-btn { flex:0 0 auto; padding:3px 6px; border:none; border-left:1px solid var(--border-color); + background:none; color:var(--fg-color); cursor:pointer; } + .mnemic-llm-arrow-btn:hover .mnemic-llm-arrow { opacity:1; } .mnemic-llm-buttons { display:flex; gap:4px; flex-wrap:wrap; } .mnemic-llm-btn { flex:1; min-width:60px; padding:3px 6px; border-radius:5px; cursor:pointer; font-size:11.5px; background:var(--comfy-input-bg); color:var(--fg-color); border:1px solid var(--border-color); white-space:nowrap; } @@ -182,9 +169,11 @@ function addStylesheet() { .mnemic-llm-item { display:flex; align-items:baseline; gap:8px; padding:4px 12px; cursor:pointer; } .mnemic-llm-item.active { background:#58a6ff33; } .mnemic-llm-item.current .mnemic-llm-item-id::before { content:"✓ "; color:#3fb950; } - .mnemic-llm-item-id { flex:1; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; } - .mnemic-llm-item-detail { opacity:.55; font-size:11px; white-space:nowrap; } - .mnemic-llm-loaded { color:#3fb950; font-size:10px; } + .mnemic-llm-item-id { flex:1; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; color:#9198a1; } + .mnemic-llm-item-tags { flex:0 0 auto; display:flex; gap:8px; font-size:11px; } + .mnemic-llm-loaded { color:#3fb950; } + .mnemic-llm-vision { color:#d29922; } + .mnemic-llm-video { color:#58a6ff; } .mnemic-llm-note { padding:6px 12px; opacity:.75; font-size:11.5px; } .mnemic-llm-note.err { color:#f85149; opacity:1; } .mnemic-llm-pop pre { margin:0; padding:10px 12px; overflow:auto; white-space:pre-wrap; word-break:break-word; font:12px/1.45 ui-monospace, monospace; } @@ -234,17 +223,18 @@ function createPopup(anchor, title) { function showModelPicker(panel, anchor) { const endpoint = panel.widget("endpoint")?.value; const pop = createPopup(anchor, `Models · ${endpoint}`); + const headTitle = pop.querySelector(".mnemic-llm-pop-head span"); const refresh = document.createElement("b"); refresh.className = "mnemic-llm-pop-refresh"; - refresh.title = "Ask the endpoint again"; + refresh.title = "Refresh: ask the endpoint for its model list again, instead of using the last one fetched (cached for 1 minute)"; refresh.textContent = "⟳"; pop.querySelector(".mnemic-llm-pop-head").insertBefore(refresh, pop.querySelector(".mnemic-llm-pop-close")); const search = document.createElement("input"); - search.placeholder = "Search, or type any model name and press Enter…"; + search.placeholder = "Search…"; const note = document.createElement("div"); note.className = "mnemic-llm-note"; - note.textContent = "Asking the endpoint…"; + note.hidden = true; const list = document.createElement("div"); list.className = "mnemic-llm-list"; pop.append(search, note, list); @@ -254,7 +244,6 @@ function showModelPicker(panel, anchor) { let shown = []; let active = 0; const current = panel.widget("model")?.value ?? ""; - const info = endpointCache.byName.get(endpoint); const choose = (id) => { panel.setModel(id); @@ -265,9 +254,9 @@ function showModelPicker(panel, anchor) { const q = search.value.trim().toLowerCase(); const terms = q.split(/\s+/).filter(Boolean); shown = models.filter((m) => terms.every((t) => `${m.id} ${m.detail ?? ""}`.toLowerCase().includes(t))); - const items = [{ id: "", label: `Endpoint default${info?.default_model ? ` (${info.default_model})` : ""}`, detail: "" }, ...shown]; + const items = [...shown]; if (q && !models.some((m) => m.id.toLowerCase() === q)) { - items.push({ id: search.value.trim(), label: `Use “${search.value.trim()}”`, detail: "custom" }); + items.push({ id: search.value.trim(), label: `Use “${search.value.trim()}”` }); } shown = items; active = Math.min(active, items.length - 1); @@ -276,24 +265,33 @@ function showModelPicker(panel, anchor) { const row = document.createElement("div"); row.className = "mnemic-llm-item"; if (i === active) row.classList.add("active"); - if (m.id === current && (m.id !== "" || current === "")) row.classList.add("current"); + if (m.id === current) row.classList.add("current"); const id = document.createElement("span"); id.className = "mnemic-llm-item-id"; id.textContent = m.label ?? m.id; id.title = m.id; row.append(id); + const tags = document.createElement("span"); + tags.className = "mnemic-llm-item-tags"; if (m.loaded) { const loaded = document.createElement("span"); loaded.className = "mnemic-llm-loaded"; - loaded.textContent = "● in memory"; - row.append(loaded); + loaded.textContent = "in memory"; + tags.append(loaded); } - if (m.detail) { - const detail = document.createElement("span"); - detail.className = "mnemic-llm-item-detail"; - detail.textContent = m.detail; - row.append(detail); + if (m.vision) { + const vision = document.createElement("span"); + vision.className = "mnemic-llm-vision"; + vision.textContent = "vision"; + tags.append(vision); } + if (m.video) { + const video = document.createElement("span"); + video.className = "mnemic-llm-video"; + video.textContent = "video"; + tags.append(video); + } + if (tags.childNodes.length) row.append(tags); row.addEventListener("mousemove", () => { if (active === i) return; list.querySelector(".active")?.classList.remove("active"); @@ -307,14 +305,16 @@ function showModelPicker(panel, anchor) { }; const load = async (force) => { - note.className = "mnemic-llm-note"; - note.textContent = "Asking the endpoint…"; const data = await getModels(endpoint, force, panel.customId()); if (openPopup !== pop) return; models = data.models ?? []; + const isStatic = models.length > 0 && models.every((m) => m.static); + headTitle.textContent = `Models · ${endpoint}${isStatic ? " *" : ""} (${models.length})`; + headTitle.title = isStatic ? "* a fixed list built into this pack, not fetched live from the endpoint" : ""; if (data.ok) { - note.textContent = `${models.length} model${models.length === 1 ? "" : "s"} · ${data.ms} ms`; + note.hidden = true; } else { + note.hidden = false; note.className = "mnemic-llm-note err"; note.textContent = `${data.error}${models.length ? " Showing models from the config instead." : ""}`; } @@ -322,7 +322,7 @@ function showModelPicker(panel, anchor) { }; search.addEventListener("input", () => { - active = search.value ? 1 : 0; + active = 0; render(); }); search.addEventListener("keydown", (e) => { @@ -344,16 +344,73 @@ function showModelPicker(panel, anchor) { load(false); } -async function showPreset(panel, anchor) { - const name = panel.widget("preset")?.value; - const pop = createPopup(anchor, name === DEFAULT_PRESET ? "No preset selected" : name); - const pre = document.createElement("pre"); - if (name === DEFAULT_PRESET) { - pre.textContent = "The node is using its own system_message field.\n\nPick a preset to replace it with a saved system prompt. Presets are shared with the Groq nodes: nodes/groq/UserPrompts.json and UserPrompts_VLM.json."; - } else { - pre.textContent = (await getPresets(name))[name] ?? "(preset not found: it may have been removed from the preset files)"; - } - pop.append(pre); +function showEndpointPicker(panel, anchor) { + const widget = panel.widget("endpoint"); + const names = widget?.options?.values ?? []; + const current = widget?.value ?? ""; + const pop = createPopup(anchor, "Endpoints"); + + const search = document.createElement("input"); + search.placeholder = "Search…"; + const list = document.createElement("div"); + list.className = "mnemic-llm-list"; + pop.append(search, list); + search.focus(); + + const choose = (name) => { + panel.setEndpoint(name); + closePopup(); + }; + + let shown = []; + let active = 0; + + const render = () => { + const q = search.value.trim().toLowerCase(); + shown = names.filter((name) => !q || name.toLowerCase().includes(q)); + active = Math.min(active, shown.length - 1); + list.replaceChildren( + ...shown.map((name, i) => { + const row = document.createElement("div"); + row.className = "mnemic-llm-item"; + if (i === active) row.classList.add("active"); + if (name === current) row.classList.add("current"); + const id = document.createElement("span"); + id.className = "mnemic-llm-item-id"; + id.textContent = name; + row.append(id); + row.addEventListener("mousemove", () => { + if (active === i) return; + list.querySelector(".active")?.classList.remove("active"); + row.classList.add("active"); + active = i; + }); + row.addEventListener("click", () => choose(name)); + return row; + }) + ); + }; + + search.addEventListener("input", () => { + active = 0; + render(); + }); + search.addEventListener("keydown", (e) => { + if (e.key === "ArrowDown" || e.key === "ArrowUp") { + e.preventDefault(); + active = (active + (e.key === "ArrowDown" ? 1 : -1) + shown.length) % shown.length; + render(); + list.querySelector(".active")?.scrollIntoView({ block: "nearest" }); + } else if (e.key === "Enter") { + e.preventDefault(); + const pick = shown[active]; + if (pick) choose(pick); + } else if (e.key === "Escape") { + closePopup(); + } + }); + + render(); } // -------------------------------------------------------------------------- @@ -372,16 +429,21 @@ class LLMPanel { - + -
- - - - +
+
+ + + + +