Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c13ee2364e | ||
|
|
b8ae298abf | ||
|
|
bb7d51777c |
@@ -173,6 +173,15 @@ python -m pip install llama-cpp-python \
|
||||
--extra-index-url https://abetlen.github.io/llama-cpp-python/whl/vulkan
|
||||
```
|
||||
|
||||
If an Apple Metal wheel is unavailable or fails archive validation, build the
|
||||
same optional requirement from source:
|
||||
|
||||
```bash
|
||||
CMAKE_ARGS="-DGGML_METAL=on" python -m pip install \
|
||||
--no-cache-dir --no-binary llama-cpp-python \
|
||||
-r ComfyUI/custom_nodes/ComfyUI_VLM_nodes/requirements-llama-cpp.txt
|
||||
```
|
||||
|
||||
The official Windows HIP Radeon index is:
|
||||
|
||||
```powershell
|
||||
|
||||
@@ -103,9 +103,10 @@ useful model capabilities:
|
||||
expression segmentation, with structured JSON, mask, and overlay outputs.
|
||||
- **PaLI-Gemma**: caption/VQA plus the official 16-token VQ-VAE segmentation
|
||||
decoder; segmentation tokens are no longer misinterpreted as polygon points.
|
||||
- **Moondream2**: pinned query API with explicit decoding controls. Its current
|
||||
checkpoint is not marked passed on the tested Torch/Transformers stack; use a
|
||||
small Modern VLM preset for production.
|
||||
- **Moondream2**: pinned query API with explicit decoding controls. The official
|
||||
checkpoint is loaded through its native safetensors state dict, avoiding the
|
||||
silent empty-output regression in Transformers 5 while retaining ComfyUI
|
||||
managed loading and unloading.
|
||||
- **Qwen2-VL**: image batches and real video-frame batches.
|
||||
- **Legacy Molmo, Kosmos-2, UForm, MCLLaVA, and MiniCPM-V 2.6 GGUF**, plus
|
||||
maintained JoyTag.
|
||||
@@ -397,6 +398,12 @@ Linux/Windows and Apple Silicon on macOS 13 or newer. It does not currently
|
||||
provide local ROCm, Intel GPU, or CPU execution. Those platforms retain every
|
||||
portable Transformers, GGUF, API, and vision utility node in this pack.
|
||||
|
||||
On CUDA 12 x86-64 systems the isolated requirements deliberately install
|
||||
`nvidia-cuda-runtime-cu12==12.9.79`. Kestrel 0.4.6's AOT kernels require the
|
||||
`cudaLibraryLoadData` entry point, which is absent from the CUDA 12.6 runtime
|
||||
bundled by cu126 PyTorch. This pin updates only Photon's private runtime; it
|
||||
does not replace ComfyUI's PyTorch build or the host NVIDIA driver.
|
||||
|
||||
GGUF nodes use optional `llama-cpp-python`. Install a wheel built for the
|
||||
desired CUDA, ROCm/HIP, Metal, Vulkan, SYCL, or CPU backend:
|
||||
|
||||
@@ -447,10 +454,13 @@ Gemma 3 and PaLI-Gemma require accepting their model licenses on Hugging Face.
|
||||
exact isolated process. `unload_after=true` gracefully shuts it down and
|
||||
terminates that process if necessary, which releases Photon model, KV-cache,
|
||||
and CUDA-graph allocations without flushing unrelated ComfyUI models. The
|
||||
sidecar intentionally does not inherit ComfyUI's PyTorch allocator override;
|
||||
Photon's CUDA-graph capture uses the native allocator in its own process. The
|
||||
worker does not inherit unrelated provider keys or proxy credentials; only
|
||||
`HF_TOKEN`, and `MOONDREAM_API_KEY` for an explicitly selected adapter, may
|
||||
cross into its server-side environment. Its random IPC secret is not placed
|
||||
on the process command line.
|
||||
cross into its server-side environment. Base-model sidecars honor
|
||||
`DO_NOT_TRACK` locally and do not start Kestrel's anonymous telemetry task.
|
||||
Its random IPC secret is not placed on the process command line.
|
||||
- A connected `video_frames` batch becomes the primary visual input. The
|
||||
optional still-image socket is ignored for video inference so smaller models
|
||||
cannot silently answer from the wrong media.
|
||||
|
||||
+68
-31
@@ -1,7 +1,21 @@
|
||||
"""Current Moondream 2 node using the model's supported query API."""
|
||||
"""Current Moondream 2 node using the model's supported query API.
|
||||
|
||||
The pinned checkpoint was authored against Transformers 4.52.4. Loading it
|
||||
through Transformers 5's ``from_pretrained`` compatibility path can silently
|
||||
produce an all-EOS model even when every tensor is reported as loaded. The
|
||||
checkpoint itself is a normal safetensors state dict, so instantiate its
|
||||
official wrapper and load that state dict directly. This keeps Moondream in
|
||||
ComfyUI's managed VRAM lifecycle without downgrading Transformers for the rest
|
||||
of the node pack.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
import torch
|
||||
|
||||
from .runtime import (
|
||||
@@ -18,12 +32,59 @@ from .runtime import (
|
||||
|
||||
MODEL_ID = "vikhyatk/moondream2"
|
||||
MODEL_REVISION = "2025-06-21"
|
||||
_CHECKPOINT_PACKAGE = "_comfyui_vlm_moondream2_checkpoint"
|
||||
|
||||
|
||||
def _checkpoint_module(model_path: str | Path):
|
||||
"""Import the checkpoint's relative modules without HF's generated cache.
|
||||
|
||||
Hugging Face's dynamic-module cache can omit transitive relative imports
|
||||
for a local snapshot. Giving the snapshot a private package namespace lets
|
||||
Python resolve the checkpoint's own ``.config``, ``.vision``, and related
|
||||
modules directly and deterministically.
|
||||
"""
|
||||
|
||||
source = str(Path(model_path).resolve())
|
||||
package = sys.modules.get(_CHECKPOINT_PACKAGE)
|
||||
if package is None:
|
||||
package = ModuleType(_CHECKPOINT_PACKAGE)
|
||||
package.__path__ = [source]
|
||||
package.__package__ = _CHECKPOINT_PACKAGE
|
||||
sys.modules[_CHECKPOINT_PACKAGE] = package
|
||||
elif list(getattr(package, "__path__", ())) != [source]:
|
||||
raise RuntimeError(
|
||||
"Moondream2 checkpoint source changed inside a running process. "
|
||||
"Restart ComfyUI before loading a different snapshot."
|
||||
)
|
||||
return importlib.import_module(f"{_CHECKPOINT_PACKAGE}.hf_moondream")
|
||||
|
||||
|
||||
def _load_native_checkpoint(model_path: str | Path):
|
||||
checkpoint = _checkpoint_module(model_path)
|
||||
safetensors = require_module("safetensors.torch")
|
||||
config = checkpoint.HfConfig.from_pretrained(
|
||||
model_path,
|
||||
local_files_only=True,
|
||||
)
|
||||
model = checkpoint.HfMoondream(config)
|
||||
weights = Path(model_path) / "model.safetensors"
|
||||
if not weights.is_file():
|
||||
raise FileNotFoundError(f"Moondream2 weights are missing: {weights}")
|
||||
missing, unexpected = safetensors.load_model(
|
||||
model,
|
||||
str(weights),
|
||||
strict=True,
|
||||
)
|
||||
if missing or unexpected:
|
||||
raise RuntimeError(
|
||||
"Moondream2 checkpoint did not load exactly: "
|
||||
f"missing={sorted(missing)}, unexpected={sorted(unexpected)}"
|
||||
)
|
||||
return model.eval()
|
||||
|
||||
|
||||
class Moondream2Predictor:
|
||||
def __init__(self):
|
||||
transformers = require_module("transformers")
|
||||
dynamic_modules = require_module("transformers.dynamic_module_utils")
|
||||
model_path = snapshot_download(
|
||||
MODEL_ID,
|
||||
"moondream2",
|
||||
@@ -31,31 +92,7 @@ class Moondream2Predictor:
|
||||
ignore_patterns=["*.bin", "*.gguf"],
|
||||
)
|
||||
self.dtype = torch_dtype("bfloat16")
|
||||
config = transformers.AutoConfig.from_pretrained(
|
||||
model_path,
|
||||
revision=MODEL_REVISION,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
remote_class = dynamic_modules.get_class_from_dynamic_module(
|
||||
"hf_moondream.HfMoondream",
|
||||
model_path,
|
||||
local_files_only=True,
|
||||
)
|
||||
|
||||
class Transformers5Moondream(remote_class):
|
||||
def __init__(self, model_config):
|
||||
super().__init__(model_config)
|
||||
# The pinned remote wrapper predates the Transformers 5 model
|
||||
# loader and does not declare its tied-weight metadata. Calling
|
||||
# the full post_init would reinitialize custom Moondream state.
|
||||
self.all_tied_weights_keys = {}
|
||||
|
||||
model = Transformers5Moondream.from_pretrained(
|
||||
model_path,
|
||||
config=config,
|
||||
dtype=self.dtype,
|
||||
)
|
||||
model.eval()
|
||||
model = _load_native_checkpoint(model_path)
|
||||
self.handle = ManagedTorchModel(model)
|
||||
|
||||
def close(self):
|
||||
@@ -92,9 +129,9 @@ class Moondream2Predictor:
|
||||
response = response.get("answer", response)
|
||||
if not str(response).strip():
|
||||
raise RuntimeError(
|
||||
"Moondream2 returned an empty response on this "
|
||||
"Torch/Transformers build. Use the Modern VLM node with "
|
||||
"LFM2.5-VL 450M, InternVL 3.5 1B, or Qwen3-VL 2B."
|
||||
"Moondream2 returned an empty response. Verify that the "
|
||||
f"{MODEL_REVISION} snapshot is complete, then restart "
|
||||
"ComfyUI so its checkpoint modules are reloaded."
|
||||
)
|
||||
results.append(str(response))
|
||||
return batch_text(results)
|
||||
|
||||
+11
-2
@@ -214,6 +214,12 @@ def _worker_environment(
|
||||
"HTTP_PROXY",
|
||||
"HTTPS_PROXY",
|
||||
"NO_PROXY",
|
||||
# ComfyUI may select cudaMallocAsync for its own allocator. Photon
|
||||
# captures CUDA graphs in a separate process, where inheriting that
|
||||
# override can abort during warmup on an uncaptured/captured free.
|
||||
# Let the sidecar use PyTorch's native allocator instead.
|
||||
"PYTORCH_ALLOC_CONF",
|
||||
"PYTORCH_CUDA_ALLOC_CONF",
|
||||
}
|
||||
secret_name = re.compile(
|
||||
r"(?:API[_-]?KEY|AUTHORIZATION|CREDENTIAL|PASSWORD|SECRET|TOKEN)",
|
||||
@@ -306,8 +312,11 @@ class Moondream31Model:
|
||||
)
|
||||
if "cudaLibraryLoadData" in tail:
|
||||
detail += (
|
||||
" The installed Photon CUDA kernel requires a newer compatible "
|
||||
"NVIDIA driver/runtime combination than this machine exposes."
|
||||
" Photon's CUDA 12 kernels require libcudart 12.9 or newer; "
|
||||
"CUDA runtime 12.6 does not export cudaLibraryLoadData. Re-run "
|
||||
"the README isolated-runtime install command so "
|
||||
"requirements-moondream31.txt upgrades only this sidecar to "
|
||||
"nvidia-cuda-runtime-cu12 12.9.79."
|
||||
)
|
||||
if tail:
|
||||
detail += f"\n\nSanitized worker log tail:\n{tail}"
|
||||
|
||||
@@ -25,6 +25,38 @@ from typing import Any
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def _honor_do_not_track() -> bool:
|
||||
"""Disable anonymous Photon reporting when the sidecar requests privacy.
|
||||
|
||||
Kestrel 0.4.2 does not currently inspect the conventional DO_NOT_TRACK
|
||||
environment variable. Base-model inference does not need its reporter, so
|
||||
keep validation local, skip the telemetry loop, and still close the HTTP
|
||||
client during engine shutdown. Finetune inference retains upstream auth
|
||||
and reporting behavior because it explicitly receives an API key.
|
||||
"""
|
||||
|
||||
if os.environ.get("DO_NOT_TRACK") != "1":
|
||||
return False
|
||||
if os.environ.get("MOONDREAM_API_KEY", "").strip():
|
||||
return False
|
||||
|
||||
from kestrel.photon import PhotonReporter
|
||||
|
||||
async def validate_api_key(self) -> bool:
|
||||
return False
|
||||
|
||||
def start(self) -> None:
|
||||
return None
|
||||
|
||||
async def shutdown(self) -> None:
|
||||
await self._client.aclose()
|
||||
|
||||
PhotonReporter.validate_api_key = validate_api_key
|
||||
PhotonReporter.start = start
|
||||
PhotonReporter.shutdown = shutdown
|
||||
return True
|
||||
|
||||
|
||||
def _register_moondream31_if_needed(model_name: str) -> bool:
|
||||
"""Bridge the official model-card ID on runtimes released before the ID.
|
||||
|
||||
@@ -283,6 +315,7 @@ def main() -> int:
|
||||
|
||||
base_model = _base_model_name(args.model)
|
||||
compatibility_registration = _register_moondream31_if_needed(base_model)
|
||||
telemetry_disabled = _honor_do_not_track()
|
||||
supported_skills = _model_skills(args.model)
|
||||
kwargs: dict[str, Any] = {
|
||||
"local": True,
|
||||
@@ -303,6 +336,7 @@ def main() -> int:
|
||||
"status": "ready",
|
||||
"moondream_version": package_version,
|
||||
"compatibility_registration": compatibility_registration,
|
||||
"telemetry_disabled": telemetry_disabled,
|
||||
"skills": sorted(supported_skills),
|
||||
"pid": os.getpid(),
|
||||
}
|
||||
|
||||
@@ -14,6 +14,7 @@ dependencies = [
|
||||
"huggingface-hub>=1.5,<2",
|
||||
"httpx>=0.27,<1",
|
||||
"jsonschema>=4.22,<5",
|
||||
"num2words>=0.5.14,<1",
|
||||
"openai>=2,<3",
|
||||
"pydantic>=2.7,<3",
|
||||
"qwen-vl-utils>=0.0.14",
|
||||
|
||||
@@ -5,3 +5,7 @@ moondream==1.3.0
|
||||
# moondream 1.3.0 expects this exact runtime API. 0.4.7+ renamed the
|
||||
# prefix-mask kernel and is not source-compatible with kestrel 0.4.2.
|
||||
kestrel-kernels==0.4.6
|
||||
# Kestrel's CUDA 12 AOT kernels call cudaLibraryLoadData. PyTorch's cu126
|
||||
# runtime (12.6.77) does not export it; 12.9.79 does and remains within the
|
||||
# CUDA 12 ABI. Keep this inside the isolated Photon environment only.
|
||||
nvidia-cuda-runtime-cu12==12.9.79; (sys_platform == "linux" and platform_machine == "x86_64") or (sys_platform == "win32" and platform_machine == "AMD64")
|
||||
|
||||
@@ -9,6 +9,7 @@ einops>=0.8,<1
|
||||
huggingface-hub>=1.5,<2
|
||||
httpx>=0.27,<1
|
||||
jsonschema>=4.22,<5
|
||||
num2words>=0.5.14,<1
|
||||
openai>=2,<3
|
||||
pydantic>=2.7,<3
|
||||
qwen-vl-utils>=0.0.14
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from types import ModuleType, SimpleNamespace
|
||||
|
||||
import torch
|
||||
|
||||
PACKAGE = __package__.split(".")[0] if __package__ else "ComfyUI_VLM_nodes"
|
||||
module = importlib.import_module(f"{PACKAGE}.nodes.moondream2")
|
||||
|
||||
|
||||
def test_native_checkpoint_loader_bypasses_transformers_from_pretrained(
|
||||
tmp_path: Path,
|
||||
monkeypatch,
|
||||
):
|
||||
package = ModuleType(module._CHECKPOINT_PACKAGE)
|
||||
package.__path__ = [str(tmp_path.resolve())]
|
||||
package.__package__ = module._CHECKPOINT_PACKAGE
|
||||
checkpoint = ModuleType(f"{module._CHECKPOINT_PACKAGE}.hf_moondream")
|
||||
calls = {}
|
||||
|
||||
class FakeConfig:
|
||||
@classmethod
|
||||
def from_pretrained(cls, model_path, **kwargs):
|
||||
calls["config"] = (Path(model_path), kwargs)
|
||||
return cls()
|
||||
|
||||
class FakeModel(torch.nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.weight = torch.nn.Parameter(torch.zeros(1))
|
||||
calls["model_config"] = config
|
||||
|
||||
checkpoint.HfConfig = FakeConfig
|
||||
checkpoint.HfMoondream = FakeModel
|
||||
monkeypatch.setitem(sys.modules, module._CHECKPOINT_PACKAGE, package)
|
||||
monkeypatch.setitem(
|
||||
sys.modules,
|
||||
f"{module._CHECKPOINT_PACKAGE}.hf_moondream",
|
||||
checkpoint,
|
||||
)
|
||||
|
||||
weights = tmp_path / "model.safetensors"
|
||||
weights.write_bytes(b"test")
|
||||
|
||||
def load_model(model, filename, *, strict):
|
||||
calls["weights"] = (model, Path(filename), strict)
|
||||
model.weight.data.fill_(1)
|
||||
return set(), []
|
||||
|
||||
monkeypatch.setattr(
|
||||
module,
|
||||
"require_module",
|
||||
lambda name: (
|
||||
SimpleNamespace(load_model=load_model)
|
||||
if name == "safetensors.torch"
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
model = module._load_native_checkpoint(tmp_path)
|
||||
|
||||
assert isinstance(model, FakeModel)
|
||||
assert not model.training
|
||||
assert model.weight.item() == 1
|
||||
assert calls["config"] == (tmp_path, {"local_files_only": True})
|
||||
assert calls["weights"] == (model, weights, True)
|
||||
|
||||
|
||||
def test_photon_requirements_pin_cuda_runtime_with_required_symbol():
|
||||
requirements = (
|
||||
Path(module.__file__).resolve().parents[1] / "requirements-moondream31.txt"
|
||||
).read_text(encoding="utf-8")
|
||||
assert "kestrel-kernels==0.4.6" in requirements
|
||||
assert "nvidia-cuda-runtime-cu12==12.9.79" in requirements
|
||||
@@ -1,3 +1,4 @@
|
||||
import asyncio
|
||||
import importlib
|
||||
import inspect
|
||||
import json
|
||||
@@ -257,6 +258,8 @@ def test_worker_auth_is_not_exposed_in_process_arguments_and_logs_are_redacted(
|
||||
monkeypatch.setenv("HF_TOKEN", "hf-server-side")
|
||||
monkeypatch.setenv("MOONDREAM_API_KEY", "adapter-only")
|
||||
monkeypatch.setenv("HTTPS_PROXY", "https://user:password@example.test")
|
||||
monkeypatch.setenv("PYTORCH_ALLOC_CONF", "backend:cudaMallocAsync")
|
||||
monkeypatch.setenv("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync")
|
||||
base_environment = module._worker_environment(
|
||||
tmp_path,
|
||||
b"\x01" * 32,
|
||||
@@ -267,6 +270,8 @@ def test_worker_auth_is_not_exposed_in_process_arguments_and_logs_are_redacted(
|
||||
assert "OPENAI_API_KEY" not in base_environment
|
||||
assert "MOONDREAM_API_KEY" not in base_environment
|
||||
assert "HTTPS_PROXY" not in base_environment
|
||||
assert "PYTORCH_ALLOC_CONF" not in base_environment
|
||||
assert "PYTORCH_CUDA_ALLOC_CONF" not in base_environment
|
||||
assert base_environment["MOONDREAM_WORKER_AUTH"] == "01" * 32
|
||||
|
||||
adapter_environment = module._worker_environment(
|
||||
@@ -325,3 +330,32 @@ def test_worker_registers_official_31_id_only_when_upstream_is_missing(
|
||||
assert registered.checkpoint_format == "md3"
|
||||
assert not worker._register_moondream31_if_needed("moondream3.1-9B-A2B")
|
||||
assert not worker._register_moondream31_if_needed("custom-model")
|
||||
|
||||
|
||||
def test_worker_honors_do_not_track_for_base_models(monkeypatch):
|
||||
class SimpleClient:
|
||||
def __init__(self):
|
||||
self.closed = False
|
||||
|
||||
async def aclose(self):
|
||||
self.closed = True
|
||||
|
||||
class Reporter:
|
||||
def __init__(self):
|
||||
self._client = SimpleClient()
|
||||
|
||||
fake = types.ModuleType("kestrel.photon")
|
||||
fake.PhotonReporter = Reporter
|
||||
monkeypatch.setitem(sys.modules, "kestrel.photon", fake)
|
||||
monkeypatch.setenv("DO_NOT_TRACK", "1")
|
||||
monkeypatch.delenv("MOONDREAM_API_KEY", raising=False)
|
||||
|
||||
assert worker._honor_do_not_track()
|
||||
reporter = Reporter()
|
||||
assert asyncio.run(reporter.validate_api_key()) is False
|
||||
assert reporter.start() is None
|
||||
asyncio.run(reporter.shutdown())
|
||||
assert reporter._client.closed
|
||||
|
||||
monkeypatch.setenv("MOONDREAM_API_KEY", "finetune-key")
|
||||
assert not worker._honor_do_not_track()
|
||||
|
||||
@@ -120,6 +120,10 @@ def test_dependency_metadata_matches_installer_requirements():
|
||||
if line.strip() and not line.lstrip().startswith("#")
|
||||
}
|
||||
assert project_requirements == installer_requirements
|
||||
assert any(
|
||||
Requirement(value).name == "num2words"
|
||||
for value in metadata["project"]["dependencies"]
|
||||
)
|
||||
|
||||
bitsandbytes = next(
|
||||
Requirement(value)
|
||||
|
||||
Reference in New Issue
Block a user