Files
WildAi d2390eb476 fix: correct realtime VibeVoice output on transformers 5.3
Garbled, parameter-insensitive speech came from a randomized EOS head:
5.3 re-runs _initialize_weights over acoustic_connector and
tts_eos_classifier because the vendored _init_weights override had no
_is_hf_initialized guard. Guard it; only checkpoint-absent weights are
initialized now.

- MockCacheLayer exposes both the 4.x and 5.x cache APIs, so the
  prefilled voice prompt is visible to 5.3's mask builder.
- _ensure_cache_has_layers covers the container: offload/prefetch,
  batch ops, crop, and a copyable lazy prefetch stream.
- max_new_tokens is a combined text+speech budget, not latents.
- cfg_scale floor of 1.5 for the realtime family.
- voice presets resolve against every registered TTS root.
- sage excluded from the realtime path: it ignores the attention mask.
- bound transformers to >=5.3.0,<5.4, the measured line.
2026-09-26 23:30:37 +03:00

81 lines
3.2 KiB
Python

"""ComfyUI folder-registration helpers for VibeVoice model assets.
The helpers accept ``folder_paths`` explicitly so registration behavior can be
unit tested without importing the package startup path or starting ComfyUI.
"""
from __future__ import annotations
import os
from typing import Any
TTS_FOLDER_KEY = "tts"
VOICE_PRESET_FOLDER_KEY = "vibevoice_voices"
VOICE_PRESET_SUBDIR = os.path.join("VibeVoice", "voices")
def _append_unique_path(paths: list[str], candidate: str) -> bool:
"""Append a normalized path once, preserving registration order."""
normalized = os.path.abspath(os.path.normpath(candidate))
existing = {os.path.abspath(os.path.normpath(path)) for path in paths}
if normalized in existing:
return False
paths.append(normalized)
return True
def register_vibevoice_folders(folder_paths_module: Any) -> list[str]:
"""Register the primary TTS root and return all registered TTS roots.
Existing ComfyUI registrations are preserved. Calling the helper more than
once does not add duplicate paths.
"""
models_dir = os.fspath(folder_paths_module.models_dir)
primary_tts_root = os.path.join(models_dir, "tts")
folder_names_and_paths = folder_paths_module.folder_names_and_paths
if TTS_FOLDER_KEY not in folder_names_and_paths:
supported_exts = set(folder_paths_module.supported_pt_extensions)
supported_exts.update({".safetensors", ".json"})
folder_names_and_paths[TTS_FOLDER_KEY] = ([], supported_exts)
registered_paths = folder_names_and_paths[TTS_FOLDER_KEY][0]
_append_unique_path(registered_paths, primary_tts_root)
return list(registered_paths)
def register_voice_preset_folder(
folder_paths_module: Any,
primary_tts_root: str,
additional_tts_roots: list[str] | None = None,
) -> list[str]:
"""Register ``<TTS root>/VibeVoice/voices`` for cached voice prompts.
Every TTS root ComfyUI knows about gets a candidate ``voices`` directory,
not just the primary one. A machine can resolve models across several roots
at once (the primary ``models/tts`` plus whatever ``extra_model_paths.yaml``
contributes), and the prompts may sit under any of them. Registering only
the first root made ``vibevoice_voices`` resolve to a directory that does
not exist, so the node reported every preset as missing even though the
prompts were installed under a different root.
Directories that do not exist are still registered: ComfyUI populates the
folder list at startup, and a root may be populated later. Discovery skips
non-existent paths, so this costs nothing.
"""
folder_names_and_paths = folder_paths_module.folder_names_and_paths
if VOICE_PRESET_FOLDER_KEY not in folder_names_and_paths:
folder_names_and_paths[VOICE_PRESET_FOLDER_KEY] = ([], {".pt"})
registered_paths = folder_names_and_paths[VOICE_PRESET_FOLDER_KEY][0]
roots = [primary_tts_root, *(additional_tts_roots or [])]
for root in roots:
if not root:
continue
voice_root = os.path.abspath(
os.path.join(os.fspath(root), VOICE_PRESET_SUBDIR)
)
_append_unique_path(registered_paths, voice_root)
return list(registered_paths)