Author SHA1 Message Date
gokayfem ad20b0eb72 Fix Moondream 2 and 3.1 local inference 2026-07-29 23:16:33 +03:00
7 changed files with 243 additions and 38 deletions
+15 -5
View File
@@ -103,9 +103,10 @@ useful model capabilities:
expression segmentation, with structured JSON, mask, and overlay outputs.
- **PaLI-Gemma**: caption/VQA plus the official 16-token VQ-VAE segmentation
decoder; segmentation tokens are no longer misinterpreted as polygon points.
- **Moondream2**: pinned query API with explicit decoding controls. Its current
checkpoint is not marked passed on the tested Torch/Transformers stack; use a
small Modern VLM preset for production.
- **Moondream2**: pinned query API with explicit decoding controls. The official
checkpoint is loaded through its native safetensors state dict, avoiding the
silent empty-output regression in Transformers 5 while retaining ComfyUI
managed loading and unloading.
- **Qwen2-VL**: image batches and real video-frame batches.
- **Legacy Molmo, Kosmos-2, UForm, MCLLaVA, and MiniCPM-V 2.6 GGUF**, plus
maintained JoyTag.
@@ -397,6 +398,12 @@ Linux/Windows and Apple Silicon on macOS 13 or newer. It does not currently
provide local ROCm, Intel GPU, or CPU execution. Those platforms retain every
portable Transformers, GGUF, API, and vision utility node in this pack.
On CUDA 12 x86-64 systems the isolated requirements deliberately install
`nvidia-cuda-runtime-cu12==12.9.79`. Kestrel 0.4.6's AOT kernels require the
`cudaLibraryLoadData` entry point, which is absent from the CUDA 12.6 runtime
bundled by cu126 PyTorch. This pin updates only Photon's private runtime; it
does not replace ComfyUI's PyTorch build or the host NVIDIA driver.
GGUF nodes use optional `llama-cpp-python`. Install a wheel built for the
desired CUDA, ROCm/HIP, Metal, Vulkan, SYCL, or CPU backend:
@@ -447,10 +454,13 @@ Gemma 3 and PaLI-Gemma require accepting their model licenses on Hugging Face.
exact isolated process. `unload_after=true` gracefully shuts it down and
terminates that process if necessary, which releases Photon model, KV-cache,
and CUDA-graph allocations without flushing unrelated ComfyUI models. The
sidecar intentionally does not inherit ComfyUI's PyTorch allocator override;
Photon's CUDA-graph capture uses the native allocator in its own process. The
worker does not inherit unrelated provider keys or proxy credentials; only
`HF_TOKEN`, and `MOONDREAM_API_KEY` for an explicitly selected adapter, may
cross into its server-side environment. Its random IPC secret is not placed
on the process command line.
cross into its server-side environment. Base-model sidecars honor
`DO_NOT_TRACK` locally and do not start Kestrel's anonymous telemetry task.
Its random IPC secret is not placed on the process command line.
- A connected `video_frames` batch becomes the primary visual input. The
optional still-image socket is ignored for video inference so smaller models
cannot silently answer from the wrong media.
+68 -31
View File
@@ -1,7 +1,21 @@
"""Current Moondream 2 node using the model's supported query API."""
"""Current Moondream 2 node using the model's supported query API.
The pinned checkpoint was authored against Transformers 4.52.4. Loading it
through Transformers 5's ``from_pretrained`` compatibility path can silently
produce an all-EOS model even when every tensor is reported as loaded. The
checkpoint itself is a normal safetensors state dict, so instantiate its
official wrapper and load that state dict directly. This keeps Moondream in
ComfyUI's managed VRAM lifecycle without downgrading Transformers for the rest
of the node pack.
"""
from __future__ import annotations
import importlib
import sys
from pathlib import Path
from types import ModuleType
import torch
from .runtime import (
@@ -18,12 +32,59 @@ from .runtime import (
MODEL_ID = "vikhyatk/moondream2"
MODEL_REVISION = "2025-06-21"
_CHECKPOINT_PACKAGE = "_comfyui_vlm_moondream2_checkpoint"
def _checkpoint_module(model_path: str | Path):
"""Import the checkpoint's relative modules without HF's generated cache.
Hugging Face's dynamic-module cache can omit transitive relative imports
for a local snapshot. Giving the snapshot a private package namespace lets
Python resolve the checkpoint's own ``.config``, ``.vision``, and related
modules directly and deterministically.
"""
source = str(Path(model_path).resolve())
package = sys.modules.get(_CHECKPOINT_PACKAGE)
if package is None:
package = ModuleType(_CHECKPOINT_PACKAGE)
package.__path__ = [source]
package.__package__ = _CHECKPOINT_PACKAGE
sys.modules[_CHECKPOINT_PACKAGE] = package
elif list(getattr(package, "__path__", ())) != [source]:
raise RuntimeError(
"Moondream2 checkpoint source changed inside a running process. "
"Restart ComfyUI before loading a different snapshot."
)
return importlib.import_module(f"{_CHECKPOINT_PACKAGE}.hf_moondream")
def _load_native_checkpoint(model_path: str | Path):
checkpoint = _checkpoint_module(model_path)
safetensors = require_module("safetensors.torch")
config = checkpoint.HfConfig.from_pretrained(
model_path,
local_files_only=True,
)
model = checkpoint.HfMoondream(config)
weights = Path(model_path) / "model.safetensors"
if not weights.is_file():
raise FileNotFoundError(f"Moondream2 weights are missing: {weights}")
missing, unexpected = safetensors.load_model(
model,
str(weights),
strict=True,
)
if missing or unexpected:
raise RuntimeError(
"Moondream2 checkpoint did not load exactly: "
f"missing={sorted(missing)}, unexpected={sorted(unexpected)}"
)
return model.eval()
class Moondream2Predictor:
def __init__(self):
transformers = require_module("transformers")
dynamic_modules = require_module("transformers.dynamic_module_utils")
model_path = snapshot_download(
MODEL_ID,
"moondream2",
@@ -31,31 +92,7 @@ class Moondream2Predictor:
ignore_patterns=["*.bin", "*.gguf"],
)
self.dtype = torch_dtype("bfloat16")
config = transformers.AutoConfig.from_pretrained(
model_path,
revision=MODEL_REVISION,
trust_remote_code=True,
)
remote_class = dynamic_modules.get_class_from_dynamic_module(
"hf_moondream.HfMoondream",
model_path,
local_files_only=True,
)
class Transformers5Moondream(remote_class):
def __init__(self, model_config):
super().__init__(model_config)
# The pinned remote wrapper predates the Transformers 5 model
# loader and does not declare its tied-weight metadata. Calling
# the full post_init would reinitialize custom Moondream state.
self.all_tied_weights_keys = {}
model = Transformers5Moondream.from_pretrained(
model_path,
config=config,
dtype=self.dtype,
)
model.eval()
model = _load_native_checkpoint(model_path)
self.handle = ManagedTorchModel(model)
def close(self):
@@ -92,9 +129,9 @@ class Moondream2Predictor:
response = response.get("answer", response)
if not str(response).strip():
raise RuntimeError(
"Moondream2 returned an empty response on this "
"Torch/Transformers build. Use the Modern VLM node with "
"LFM2.5-VL 450M, InternVL 3.5 1B, or Qwen3-VL 2B."
"Moondream2 returned an empty response. Verify that the "
f"{MODEL_REVISION} snapshot is complete, then restart "
"ComfyUI so its checkpoint modules are reloaded."
)
results.append(str(response))
return batch_text(results)
+11 -2
View File
@@ -214,6 +214,12 @@ def _worker_environment(
"HTTP_PROXY",
"HTTPS_PROXY",
"NO_PROXY",
# ComfyUI may select cudaMallocAsync for its own allocator. Photon
# captures CUDA graphs in a separate process, where inheriting that
# override can abort during warmup on an uncaptured/captured free.
# Let the sidecar use PyTorch's native allocator instead.
"PYTORCH_ALLOC_CONF",
"PYTORCH_CUDA_ALLOC_CONF",
}
secret_name = re.compile(
r"(?:API[_-]?KEY|AUTHORIZATION|CREDENTIAL|PASSWORD|SECRET|TOKEN)",
@@ -306,8 +312,11 @@ class Moondream31Model:
)
if "cudaLibraryLoadData" in tail:
detail += (
" The installed Photon CUDA kernel requires a newer compatible "
"NVIDIA driver/runtime combination than this machine exposes."
" Photon's CUDA 12 kernels require libcudart 12.9 or newer; "
"CUDA runtime 12.6 does not export cudaLibraryLoadData. Re-run "
"the README isolated-runtime install command so "
"requirements-moondream31.txt upgrades only this sidecar to "
"nvidia-cuda-runtime-cu12 12.9.79."
)
if tail:
detail += f"\n\nSanitized worker log tail:\n{tail}"
+34
View File
@@ -25,6 +25,38 @@ from typing import Any
from PIL import Image
def _honor_do_not_track() -> bool:
"""Disable anonymous Photon reporting when the sidecar requests privacy.
Kestrel 0.4.2 does not currently inspect the conventional DO_NOT_TRACK
environment variable. Base-model inference does not need its reporter, so
keep validation local, skip the telemetry loop, and still close the HTTP
client during engine shutdown. Finetune inference retains upstream auth
and reporting behavior because it explicitly receives an API key.
"""
if os.environ.get("DO_NOT_TRACK") != "1":
return False
if os.environ.get("MOONDREAM_API_KEY", "").strip():
return False
from kestrel.photon import PhotonReporter
async def validate_api_key(self) -> bool:
return False
def start(self) -> None:
return None
async def shutdown(self) -> None:
await self._client.aclose()
PhotonReporter.validate_api_key = validate_api_key
PhotonReporter.start = start
PhotonReporter.shutdown = shutdown
return True
def _register_moondream31_if_needed(model_name: str) -> bool:
"""Bridge the official model-card ID on runtimes released before the ID.
@@ -283,6 +315,7 @@ def main() -> int:
base_model = _base_model_name(args.model)
compatibility_registration = _register_moondream31_if_needed(base_model)
telemetry_disabled = _honor_do_not_track()
supported_skills = _model_skills(args.model)
kwargs: dict[str, Any] = {
"local": True,
@@ -303,6 +336,7 @@ def main() -> int:
"status": "ready",
"moondream_version": package_version,
"compatibility_registration": compatibility_registration,
"telemetry_disabled": telemetry_disabled,
"skills": sorted(supported_skills),
"pid": os.getpid(),
}
+4
View File
@@ -5,3 +5,7 @@ moondream==1.3.0
# moondream 1.3.0 expects this exact runtime API. 0.4.7+ renamed the
# prefix-mask kernel and is not source-compatible with kestrel 0.4.2.
kestrel-kernels==0.4.6
# Kestrel's CUDA 12 AOT kernels call cudaLibraryLoadData. PyTorch's cu126
# runtime (12.6.77) does not export it; 12.9.79 does and remains within the
# CUDA 12 ABI. Keep this inside the isolated Photon environment only.
nvidia-cuda-runtime-cu12==12.9.79; (sys_platform == "linux" and platform_machine == "x86_64") or (sys_platform == "win32" and platform_machine == "AMD64")
+77
View File
@@ -0,0 +1,77 @@
from __future__ import annotations
import importlib
import sys
from pathlib import Path
from types import ModuleType, SimpleNamespace
import torch
PACKAGE = __package__.split(".")[0] if __package__ else "ComfyUI_VLM_nodes"
module = importlib.import_module(f"{PACKAGE}.nodes.moondream2")
def test_native_checkpoint_loader_bypasses_transformers_from_pretrained(
tmp_path: Path,
monkeypatch,
):
package = ModuleType(module._CHECKPOINT_PACKAGE)
package.__path__ = [str(tmp_path.resolve())]
package.__package__ = module._CHECKPOINT_PACKAGE
checkpoint = ModuleType(f"{module._CHECKPOINT_PACKAGE}.hf_moondream")
calls = {}
class FakeConfig:
@classmethod
def from_pretrained(cls, model_path, **kwargs):
calls["config"] = (Path(model_path), kwargs)
return cls()
class FakeModel(torch.nn.Module):
def __init__(self, config):
super().__init__()
self.weight = torch.nn.Parameter(torch.zeros(1))
calls["model_config"] = config
checkpoint.HfConfig = FakeConfig
checkpoint.HfMoondream = FakeModel
monkeypatch.setitem(sys.modules, module._CHECKPOINT_PACKAGE, package)
monkeypatch.setitem(
sys.modules,
f"{module._CHECKPOINT_PACKAGE}.hf_moondream",
checkpoint,
)
weights = tmp_path / "model.safetensors"
weights.write_bytes(b"test")
def load_model(model, filename, *, strict):
calls["weights"] = (model, Path(filename), strict)
model.weight.data.fill_(1)
return set(), []
monkeypatch.setattr(
module,
"require_module",
lambda name: (
SimpleNamespace(load_model=load_model)
if name == "safetensors.torch"
else None
),
)
model = module._load_native_checkpoint(tmp_path)
assert isinstance(model, FakeModel)
assert not model.training
assert model.weight.item() == 1
assert calls["config"] == (tmp_path, {"local_files_only": True})
assert calls["weights"] == (model, weights, True)
def test_photon_requirements_pin_cuda_runtime_with_required_symbol():
requirements = (
Path(module.__file__).resolve().parents[1] / "requirements-moondream31.txt"
).read_text(encoding="utf-8")
assert "kestrel-kernels==0.4.6" in requirements
assert "nvidia-cuda-runtime-cu12==12.9.79" in requirements
+34
View File
@@ -1,3 +1,4 @@
import asyncio
import importlib
import inspect
import json
@@ -257,6 +258,8 @@ def test_worker_auth_is_not_exposed_in_process_arguments_and_logs_are_redacted(
monkeypatch.setenv("HF_TOKEN", "hf-server-side")
monkeypatch.setenv("MOONDREAM_API_KEY", "adapter-only")
monkeypatch.setenv("HTTPS_PROXY", "https://user:password@example.test")
monkeypatch.setenv("PYTORCH_ALLOC_CONF", "backend:cudaMallocAsync")
monkeypatch.setenv("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync")
base_environment = module._worker_environment(
tmp_path,
b"\x01" * 32,
@@ -267,6 +270,8 @@ def test_worker_auth_is_not_exposed_in_process_arguments_and_logs_are_redacted(
assert "OPENAI_API_KEY" not in base_environment
assert "MOONDREAM_API_KEY" not in base_environment
assert "HTTPS_PROXY" not in base_environment
assert "PYTORCH_ALLOC_CONF" not in base_environment
assert "PYTORCH_CUDA_ALLOC_CONF" not in base_environment
assert base_environment["MOONDREAM_WORKER_AUTH"] == "01" * 32
adapter_environment = module._worker_environment(
@@ -325,3 +330,32 @@ def test_worker_registers_official_31_id_only_when_upstream_is_missing(
assert registered.checkpoint_format == "md3"
assert not worker._register_moondream31_if_needed("moondream3.1-9B-A2B")
assert not worker._register_moondream31_if_needed("custom-model")
def test_worker_honors_do_not_track_for_base_models(monkeypatch):
class SimpleClient:
def __init__(self):
self.closed = False
async def aclose(self):
self.closed = True
class Reporter:
def __init__(self):
self._client = SimpleClient()
fake = types.ModuleType("kestrel.photon")
fake.PhotonReporter = Reporter
monkeypatch.setitem(sys.modules, "kestrel.photon", fake)
monkeypatch.setenv("DO_NOT_TRACK", "1")
monkeypatch.delenv("MOONDREAM_API_KEY", raising=False)
assert worker._honor_do_not_track()
reporter = Reporter()
assert asyncio.run(reporter.validate_api_key()) is False
assert reporter.start() is None
asyncio.run(reporter.shutdown())
assert reporter._client.closed
monkeypatch.setenv("MOONDREAM_API_KEY", "finetune-key")
assert not worker._honor_do_not_track()