Compare commits

...
Author SHA1 Message Date
diodiogod 5a28a46010 Add audio.cpp multi-task support 2026-08-13 21:41:16 -03:00
diodiogod 211b192f4a Improve audio.cpp console feedback 2026-08-12 18:04:27 -03:00
diodiogod 4a99f15851 Add audio.cpp TTS engine integration 2026-08-12 11:50:10 -03:00
73 changed files with 15332 additions and 36 deletions
+2
View File
@@ -355,6 +355,8 @@ def setup_api_routes():
from utils.voice.alias_api import register_character_alias_routes
register_character_alias_routes(PromptServer.instance.routes, web)
from utils.audio_cpp.capability_api import register_audio_cpp_capability_routes
register_audio_cpp_capability_routes(PromptServer.instance.routes, web)
@PromptServer.instance.routes.get("/api/tts-audio-suite/index-tts-emotion-presets")
async def get_index_tts_emotion_presets_endpoint(request):
+486
View File
@@ -0,0 +1,486 @@
"""audio.cpp adapter for the Suite's unified ASR pipeline."""
from __future__ import annotations
import json
import os
import re
import time
from typing import Any, Dict, Iterable, Mapping, Optional
import torch
from utils.asr.types import ASRRequest, ASRResult, ASRSegment, ASRWord
from utils.audio.processing import AudioProcessingUtils
_NATIVE_CHUNK_FAMILIES = {
"fun_asr_nano",
"higgs_audio_stt",
"hviske_asr",
"qwen3_asr",
"vibevoice_asr",
"voxtral_realtime",
}
# audio.cpp release-0.5.1 keeps stale offline decoder state for these loaders:
# the first request transcribes normally and later requests return empty text.
# A fresh owned process is currently the only reliable reset contract.
_RESTART_BETWEEN_CHUNKS_FAMILIES = {"nemotron_asr", "voxtral_realtime"}
def _session(config: Mapping[str, Any]):
from utils.audio_cpp.session import get_audio_cpp_session
return get_audio_cpp_session(dict(config))
def _advanced_options(config: Mapping[str, Any]) -> Dict[str, Any]:
value = config.get("advanced_options", config.get("request_options", {}))
if value in (None, ""):
return {}
if isinstance(value, str):
try:
value = json.loads(value)
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
if not isinstance(value, Mapping):
raise ValueError("audio.cpp advanced options must be a JSON object")
return dict(value)
def _audio_path(audio: Mapping[str, Any]) -> str:
waveform = audio.get("waveform")
sample_rate = audio.get("sample_rate")
if not torch.is_tensor(waveform):
raise TypeError("audio.cpp ASR input must contain a waveform tensor")
if sample_rate is None or int(sample_rate) <= 0:
raise ValueError("audio.cpp ASR input must contain a positive sample_rate")
return os.path.abspath(
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
)
def _waveform_3d(audio: Mapping[str, Any]) -> tuple[torch.Tensor, int]:
waveform = audio.get("waveform")
sample_rate = int(audio.get("sample_rate") or 0)
if not torch.is_tensor(waveform):
raise TypeError("audio.cpp ASR input must contain a waveform tensor")
if sample_rate <= 0:
raise ValueError("audio.cpp ASR input must contain a positive sample_rate")
if waveform.ndim == 1:
waveform = waveform.unsqueeze(0).unsqueeze(0)
elif waveform.ndim == 2:
waveform = waveform.unsqueeze(0)
elif waveform.ndim != 3:
raise ValueError(
"audio.cpp ASR waveform must have [samples], [channels, samples], or "
"[batch, channels, samples] shape"
)
if waveform.shape[0] != 1:
raise ValueError("audio.cpp ASR accepts one audio item at a time")
if waveform.shape[-1] <= 0:
raise ValueError("audio.cpp ASR input audio is empty")
return waveform.detach().cpu(), sample_rate
def _chunk_ranges(
total_samples: int,
sample_rate: int,
chunk_size: int,
overlap: int,
) -> list[tuple[int, int]]:
if chunk_size <= 0:
return [(0, total_samples)]
if overlap < 0:
raise ValueError("ASR overlap must be zero or greater")
if overlap >= chunk_size:
raise ValueError("ASR overlap must be smaller than chunk_size")
chunk_samples = chunk_size * sample_rate
if total_samples <= chunk_samples:
return [(0, total_samples)]
step_samples = (chunk_size - overlap) * sample_rate
ranges = []
start = 0
while start < total_samples:
end = min(start + chunk_samples, total_samples)
ranges.append((start, end))
if end >= total_samples:
break
start += step_samples
return ranges
def _normalized_token(value: str) -> str:
return re.sub(r"[^\w]+", "", value, flags=re.UNICODE).casefold()
def _merge_transcript(parts: Iterable[str]) -> str:
merged: list[str] = []
for part in parts:
incoming = str(part or "").strip().split()
if not incoming:
continue
if not merged:
merged.extend(incoming)
continue
limit = min(len(merged), len(incoming), 80)
duplicate_count = 0
for size in range(limit, 0, -1):
left = [_normalized_token(token) for token in merged[-size:]]
right = [_normalized_token(token) for token in incoming[:size]]
if all(left) and left == right:
duplicate_count = size
break
merged.extend(incoming[duplicate_count:])
return " ".join(merged).strip()
def _offset_words(
words: Iterable[ASRWord], offset: float, unique_after: Optional[float]
) -> list[ASRWord]:
shifted = []
for word in words:
item = ASRWord(start=word.start + offset, end=word.end + offset, text=word.text)
if unique_after is not None and (item.start + item.end) / 2.0 < unique_after:
continue
shifted.append(item)
return shifted
def _offset_segments(
segments: Iterable[ASRSegment], offset: float, unique_after: Optional[float]
) -> list[ASRSegment]:
shifted = []
for segment in segments:
item = ASRSegment(
start=segment.start + offset,
end=segment.end + offset,
text=segment.text,
speaker=segment.speaker,
)
if unique_after is not None and (item.start + item.end) / 2.0 < unique_after:
continue
shifted.append(item)
return shifted
def _seconds(value: Any, sample_rate: int) -> float:
try:
return max(0.0, float(value) / float(sample_rate))
except (TypeError, ValueError, ZeroDivisionError):
return 0.0
def _words(payload: Mapping[str, Any], sample_rate: int) -> list[ASRWord]:
words = []
for item in payload.get("words") or []:
if not isinstance(item, Mapping):
continue
text = str(item.get("word", item.get("text", ""))).strip()
if not text:
continue
words.append(
ASRWord(
start=_seconds(item.get("start_sample"), sample_rate),
end=_seconds(item.get("end_sample"), sample_rate),
text=text,
)
)
return words
def _plain_segments(payload: Mapping[str, Any], sample_rate: int) -> list[ASRSegment]:
segments = []
for item in payload.get("segments") or []:
if not isinstance(item, Mapping):
continue
text = str(item.get("text", "")).strip()
segments.append(
ASRSegment(
start=_seconds(item.get("start_sample"), sample_rate),
end=_seconds(item.get("end_sample"), sample_rate),
text=text,
)
)
return segments
def _speaker_segments(payload: Mapping[str, Any], sample_rate: int) -> list[ASRSegment]:
segments = []
for item in payload.get("speaker_turns") or []:
if not isinstance(item, Mapping):
continue
speaker = str(item.get("speaker_id", "")).strip()
if speaker and not speaker.lower().startswith("speaker"):
speaker = f"Speaker {speaker}"
segments.append(
ASRSegment(
start=_seconds(item.get("start_sample"), sample_rate),
end=_seconds(item.get("end_sample"), sample_rate),
text=str(item.get("text", "")).strip(),
speaker=speaker or None,
)
)
return segments
def _attach_words(segments: Iterable[ASRSegment], words: Iterable[ASRWord]) -> None:
segment_list = list(segments)
for word in words:
midpoint = (word.start + word.end) / 2.0
target = next(
(segment for segment in segment_list if segment.start <= midpoint <= segment.end),
None,
)
if target is not None:
target.words.append(word)
class AudioCppASREngineAdapter:
"""Normalize audio.cpp transcript/timing output into ``ASRResult``."""
def __init__(self, engine_data: Dict[str, Any]):
self.engine_data = dict(engine_data)
self.config = dict(engine_data.get("config", engine_data))
def _session_config(self) -> Dict[str, Any]:
config = dict(self.config)
if str(config.get("connection_mode", "auto")).lower() != "external_server":
config["requested_task"] = "asr"
config["task"] = "asr"
return config
def transcribe(self, req: ASRRequest) -> ASRResult:
if req.task != "transcribe":
raise ValueError(
"audio.cpp release-0.5.1 ASR loaders support transcription, not the "
"Unified ASR translate mode"
)
config = self._session_config()
family = str(config.get("family", "")).strip()
warnings: list[str] = []
notes: list[str] = []
options = _advanced_options(config)
# VibeVoice-ASR owns diarization across its full recording. Independent
# Suite requests can restart speaker numbering, so preserve its native
# chunking only for this mode. All other ASR uses Suite-side windows.
native_diarization = (
family == "vibevoice_asr" and req.diarization and req.chunk_size > 0
)
if native_diarization:
options.setdefault("audio_chunk_mode", "fixed")
options.setdefault("audio_chunk_seconds", int(req.chunk_size))
if req.overlap > 0:
notes.append(
"VibeVoice-ASR diarization uses native chunking to preserve speaker "
"identity; the Suite overlap setting is not applied."
)
elif family in _NATIVE_CHUNK_FAMILIES:
options.setdefault("audio_chunk_mode", "none")
if req.timestamps == "word" and family == "qwen3_asr":
session_options = config.get("session_options") or {}
aligner = session_options.get("qwen3_asr.forced_aligner_model_path")
if aligner:
options["return_timestamps"] = True
else:
warnings.append(
"Qwen3-ASR word timestamps require the optional Qwen3 Forced Aligner; "
"transcription continued without downloading that auxiliary model."
)
waveform, source_rate = _waveform_3d(req.audio)
ranges = (
[(0, waveform.shape[-1])]
if native_diarization
else _chunk_ranges(
waveform.shape[-1], source_rate, int(req.chunk_size), int(req.overlap)
)
)
session = _session(config)
if str(getattr(session, "task", "asr")) != "asr":
raise ValueError(
f"audio.cpp model '{session.model_id}' is configured for task "
f"'{session.task}', not ASR"
)
restart_between_chunks = (
len(ranges) > 1 and family in _RESTART_BETWEEN_CHUNKS_FAMILIES
)
if restart_between_chunks and not bool(getattr(session, "owned", False)):
raise RuntimeError(
f"audio.cpp release-0.5.1 {family} returns empty text after its first "
"offline request. Suite-side chunking therefore requires a managed "
"audio.cpp server so the Suite can reset it between chunks. Set "
"connection_mode to managed, or set ASR chunk_size to 0 when using "
"an external server."
)
if restart_between_chunks:
notes.append(
f"audio.cpp release-0.5.1 {family} requires a managed server reset "
"between Suite chunks to avoid empty repeated-request results."
)
display_family = family or "external model"
print(f"🎧 audio.cpp ASR: Transcribing with {display_family}...")
if len(ranges) > 1:
notes.append(
f"Suite-side ASR chunking used {len(ranges)} windows of "
f"{int(req.chunk_size)}s with {int(req.overlap)}s overlap."
)
print(
f"🧩 audio.cpp ASR: {len(ranges)} chunks "
f"({int(req.chunk_size)}s, {int(req.overlap)}s overlap)"
)
payloads: list[Mapping[str, Any]] = []
chunk_timings: list[Mapping[str, Any]] = []
chunk_diagnostics: list[Dict[str, Any]] = []
started_at = time.time()
for index, (start, end) in enumerate(ranges, start=1):
if index > 1 and restart_between_chunks:
print(
f"🔄 audio.cpp ASR: Resetting {family} session for chunk "
f"{index}/{len(ranges)}"
)
session.restart_owned_runtime()
chunk_waveform = waveform[..., start:end]
chunk_rms = float(torch.sqrt(torch.mean(chunk_waveform.float().square())).item())
chunk_peak = float(chunk_waveform.float().abs().max().item())
temp_path = _audio_path({
"waveform": chunk_waveform,
"sample_rate": source_rate,
})
try:
request: Dict[str, Any] = {"audio": temp_path, "options": dict(options)}
if req.language:
request["language"] = req.language
result = session.run(request)
payload = result.raw if isinstance(result.raw, Mapping) else {}
payloads.append(payload)
if isinstance(payload.get("timing"), Mapping):
chunk_timings.append(payload["timing"])
chunk_diagnostics.append({
"index": index,
"start": round(start / source_rate, 3),
"end": round(end / source_rate, 3),
"rms": round(chunk_rms, 6),
"peak": round(chunk_peak, 6),
"text": str(payload.get("text", "")).strip(),
"characters": len(str(payload.get("text", "")).strip()),
"upstream_timing": (
dict(payload["timing"])
if isinstance(payload.get("timing"), Mapping)
else None
),
})
finally:
try:
os.remove(temp_path)
except FileNotFoundError:
pass
if len(ranges) > 1:
chunk_chars = len(str(payload.get("text", "")).strip())
print(
f" ASR chunk {index}/{len(ranges)} complete "
f"({chunk_chars} chars, RMS {chunk_rms:.4f}, peak {chunk_peak:.4f})"
)
words: list[ASRWord] = []
speaker_segments: list[ASRSegment] = []
plain_segments: list[ASRSegment] = []
overlap_seconds = float(req.overlap) if len(ranges) > 1 else 0.0
for index, ((start, _end), payload) in enumerate(zip(ranges, payloads)):
offset = start / source_rate
unique_after = offset + overlap_seconds if index > 0 else None
words.extend(_offset_words(_words(payload, source_rate), offset, unique_after))
speaker_segments.extend(
_offset_segments(
_speaker_segments(payload, source_rate), offset, unique_after
)
)
plain_segments.extend(
_offset_segments(_plain_segments(payload, source_rate), offset, unique_after)
)
if req.diarization:
segments = speaker_segments
if segments:
_attach_words(segments, words)
else:
warnings.append(
f"audio.cpp {family or 'ASR model'} returned no speaker-attributed turns."
)
segments = plain_segments
elif req.timestamps == "word" and words:
segments = [
ASRSegment(start=word.start, end=word.end, text=word.text, words=[word])
for word in words
]
elif req.timestamps == "word":
segments = plain_segments
else:
segments = []
text = _merge_transcript(payload.get("text", "") for payload in payloads)
if req.diarization and speaker_segments:
text = " ".join(
f"[{segment.speaker}] {segment.text}" if segment.speaker else segment.text
for segment in speaker_segments
if segment.text
).strip()
if not text and speaker_segments:
text = " ".join(segment.text for segment in speaker_segments if segment.text).strip()
if req.timestamps == "word" and not words:
warnings.append(f"audio.cpp {family or 'ASR model'} returned no word timestamps.")
empty_chunks = sum(
1 for payload in payloads if not str(payload.get("text", "")).strip()
)
if len(payloads) > 1 and empty_chunks:
warnings.append(
f"audio.cpp {family or 'ASR model'} returned no text for "
f"{empty_chunks} of {len(payloads)} Suite chunks."
)
raw: Dict[str, Any] = {}
if warnings:
raw["warnings"] = warnings
if notes:
raw["notes"] = notes
if len(payloads) == 1 and chunk_timings:
raw["timing"] = dict(chunk_timings[0])
elif len(payloads) > 1:
raw["timing"] = {
"wall_ms": round((time.time() - started_at) * 1000.0, 3),
"suite_chunks": len(payloads),
"suite_chunk_size_seconds": int(req.chunk_size),
"suite_overlap_seconds": int(req.overlap),
"upstream_wall_ms": round(
sum(float(item.get("wall_ms", 0.0)) for item in chunk_timings), 3
),
}
raw["chunks"] = chunk_diagnostics
output_language = next(
(
str(payload.get("language", "")).strip()
for payload in payloads
if str(payload.get("language", "")).strip()
),
str(req.language or "").strip(),
) or None
print(
f"✅ audio.cpp ASR: Complete ({len(text)} chars, "
f"{len(segments)} timed/speaker segments)"
)
return ASRResult(
text=text,
language=output_language,
segments=segments,
raw=raw or None,
)
__all__ = ["AudioCppASREngineAdapter"]
+372
View File
@@ -0,0 +1,372 @@
"""Adapter between the suite's TTS processors and an audio.cpp session."""
from __future__ import annotations
import json
import os
import threading
from typing import Any, Dict, Mapping, Optional, Tuple
import torch
from utils.audio.audio_hash import generate_stable_audio_component
from utils.audio.cache import get_audio_cache
from utils.audio.processing import AudioProcessingUtils
from utils.voice.reference import effective_voice_audio
_CACHE_SAMPLE_RATES: Dict[str, int] = {}
_CACHE_SAMPLE_RATES_LOCK = threading.Lock()
def _get_session(config: Mapping[str, Any]):
"""Import lazily so the node can still be discovered before optional setup."""
from utils.audio_cpp.session import get_audio_cpp_session
return get_audio_cpp_session(dict(config))
def _canonical_json(value: Mapping[str, Any]) -> str:
return json.dumps(value, sort_keys=True, separators=(",", ":"), default=str)
class AudioCppEngineAdapter:
"""Build generic ``/v1/tasks/run`` requests and retain their real sample rate."""
_COMMON_REQUEST_FIELDS = (
"temperature",
"top_p",
"top_k",
"repetition_penalty",
"max_tokens",
"max_steps",
"num_inference_steps",
"guidance_scale",
"speaking_rate",
)
def __init__(self, config: Optional[Dict[str, Any]] = None):
self.config = dict(config or {})
self.audio_cache = get_audio_cache()
self._last_sample_rate: Optional[int] = None
self._reference_files: Dict[str, str] = {}
self._reference_lock = threading.RLock()
@property
def sample_rate(self) -> Optional[int]:
return self._last_sample_rate
def update_config(self, new_config: Optional[Dict[str, Any]]) -> None:
self.config = dict(new_config or {})
@staticmethod
def _reference_text(voice_ref: Any) -> str:
if not isinstance(voice_ref, Mapping):
return ""
return str(
voice_ref.get("reference_text")
or voice_ref.get("prompt_text")
or voice_ref.get("text")
or ""
).strip()
def _materialize_reference(self, voice_ref: Any) -> Tuple[Optional[str], str, str, Optional[str]]:
"""Return path, transcript, stable hash, and the path that must be removed."""
reference_text = AudioCppEngineAdapter._reference_text(voice_ref)
if not isinstance(voice_ref, Mapping):
return None, reference_text, "default_voice", None
audio = effective_voice_audio(voice_ref)
if audio is None:
return None, reference_text, "default_voice", None
if isinstance(audio, (str, os.PathLike)):
path = os.path.abspath(os.path.expanduser(os.fspath(audio)))
if not os.path.isfile(path):
raise FileNotFoundError(f"audio.cpp reference audio not found: {path}")
component = generate_stable_audio_component(audio_file_path=path)
return path, reference_text, component, None
if isinstance(audio, Mapping):
waveform = audio.get("waveform")
sample_rate = audio.get("sample_rate")
audio_dict = dict(audio)
elif torch.is_tensor(audio):
waveform = audio
sample_rate = voice_ref.get("sample_rate")
audio_dict = {"waveform": waveform, "sample_rate": sample_rate}
else:
raise TypeError(f"Unsupported audio.cpp voice reference type: {type(audio).__name__}")
if not torch.is_tensor(waveform):
raise TypeError("audio.cpp reference audio must contain a waveform tensor")
if sample_rate is None or int(sample_rate) <= 0:
raise ValueError("audio.cpp reference audio must contain a positive sample_rate")
audio_dict["sample_rate"] = int(sample_rate)
component = generate_stable_audio_component(reference_audio=audio_dict)
if component not in {"ref_audio_error", "ref_audio_error_not_tensor"}:
with self._reference_lock:
cached_path = self._reference_files.get(component)
if cached_path and os.path.isfile(cached_path):
return cached_path, reference_text, component, None
temp_path = os.path.abspath(
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
)
self._reference_files[component] = temp_path
return temp_path, reference_text, component, None
# Hash failures must not make unrelated references share one file.
temp_path = os.path.abspath(
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
)
return temp_path, reference_text, component, temp_path
def close(self) -> None:
with self._reference_lock:
paths = list(self._reference_files.values())
self._reference_files.clear()
for path in paths:
try:
os.remove(path)
except FileNotFoundError:
pass
except OSError:
pass
def __del__(self):
try:
self.close()
except Exception:
pass
def _advanced_options(self) -> Dict[str, Any]:
value = self.config.get(
"advanced_options",
self.config.get("request_options", self.config.get("advanced_json", {})),
)
if value in (None, ""):
return {}
if isinstance(value, str):
try:
value = json.loads(value)
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
if not isinstance(value, Mapping):
raise ValueError("audio.cpp advanced options must be a JSON object")
return dict(value)
def _resolved_task(self, session: Any) -> str:
requested = str(self.config.get("task", self.config.get("requested_task", "auto"))).lower()
for source in (session, getattr(session, "config", None)):
if source is None:
continue
value = source.get("task") if isinstance(source, Mapping) else getattr(source, "task", None)
if str(value).lower() in {"tts", "clon", "vdes"}:
return str(value).lower()
if requested in {"tts", "clon", "vdes"}:
return requested
if str(self.config.get("connection_mode", "auto")).lower() == "external_server":
return "auto"
try:
from utils.audio_cpp.catalog import resolve_task
return str(
resolve_task(
self.config.get("family", ""),
self.config.get("package_id", ""),
requested="auto",
)
).lower()
except (ImportError, KeyError, TypeError, ValueError):
return "tts"
def _build_request(
self,
text: str,
voice_path: Optional[str],
reference_text: str,
seed: int,
advanced: Dict[str, Any],
task: str,
) -> Dict[str, Any]:
request: Dict[str, Any] = {"text": text, "seed": str(int(seed)), "options": advanced}
del task # The persistent session owns its one configured model/task.
language = str(self.config.get("language", "")).strip()
if language and language.lower() not in {"auto", "none"}:
request["language"] = language
voice_id = str(self.config.get("voice_id", self.config.get("voice", ""))).strip()
if voice_id:
request["voice_id"] = voice_id
if voice_path:
request["voice_ref"] = voice_path
if reference_text:
request["reference_text"] = reference_text
instruct = str(self.config.get("instruct", "")).strip()
if instruct:
request["instruct"] = instruct
for key in self._COMMON_REQUEST_FIELDS:
value = self.config.get(key)
if value is not None and value != "":
request[key] = value
return request
def _cache_key(
self,
text: str,
audio_component: str,
reference_text: str,
seed: int,
task: str,
advanced: Dict[str, Any],
character_name: Optional[str],
session: Any,
) -> str:
session_config = getattr(session, "config", {})
if not isinstance(session_config, Mapping):
session_config = {}
session_family = getattr(session, "family", None) or session_config.get(
"family", self.config.get("family", "")
)
session_model_id = getattr(session, "model_id", None) or session_config.get(
"model_id", self.config.get("model_id", "")
)
# Owned servers use a random loopback port on every restart; that port is
# transport state, not model identity. External endpoints are stable and
# must participate in the cache key.
if bool(getattr(session, "owned", False)):
session_endpoint = ""
else:
session_endpoint = getattr(session, "endpoint", None) or self.config.get(
"server_url", self.config.get("external_server_url", "")
)
extra_identity = {
"options": advanced,
"speaking_rate": self.config.get("speaking_rate"),
"connection_mode": self.config.get("connection_mode", "auto"),
"server_url": session_endpoint,
"binary_path": session_config.get("binary_path", self.config.get("binary_path", "")),
"backend": session_config.get("backend", self.config.get("backend", "")),
"device": session_config.get("device", self.config.get("device", "")),
"load_options": session_config.get("load_options", self.config.get("load_options", {})),
"session_options": session_config.get(
"session_options", self.config.get("session_options", {})
),
"default_request_options": session_config.get(
"default_request_options", self.config.get("default_request_options", {})
),
}
return self.audio_cache.generate_cache_key(
"audio_cpp",
text=text,
audio_component=audio_component,
reference_text=reference_text,
family=session_family,
package_id=session_config.get("package_id", self.config.get("package_id", "")),
model_path=session_config.get("model_path", self.config.get("model_path", "")),
model_id=session_model_id,
task=task,
language=self.config.get("language", ""),
voice_id=self.config.get("voice_id", self.config.get("voice", "")),
instruct=self.config.get("instruct", ""),
temperature=self.config.get("temperature"),
top_p=self.config.get("top_p"),
top_k=self.config.get("top_k"),
repetition_penalty=self.config.get("repetition_penalty"),
max_tokens=self.config.get("max_tokens"),
max_steps=self.config.get("max_steps"),
num_inference_steps=self.config.get("num_inference_steps"),
guidance_scale=self.config.get("guidance_scale"),
seed=int(seed),
request_options=_canonical_json(extra_identity),
character=character_name or "narrator",
)
@staticmethod
def _normalize_result(result: Any) -> Tuple[torch.Tensor, int]:
waveform = result.get("waveform") if isinstance(result, Mapping) else getattr(result, "waveform", None)
sample_rate = result.get("sample_rate") if isinstance(result, Mapping) else getattr(result, "sample_rate", None)
if waveform is None:
named = result.get("named_audio", {}) if isinstance(result, Mapping) else getattr(result, "named_audio", {})
values = list(named.values()) if isinstance(named, Mapping) else list(named or [])
if len(values) == 1:
item = values[0]
waveform = item.get("waveform") if isinstance(item, Mapping) else getattr(item, "waveform", None)
sample_rate = sample_rate or (item.get("sample_rate") if isinstance(item, Mapping) else getattr(item, "sample_rate", None))
if waveform is None:
raise RuntimeError("audio.cpp returned no primary audio output")
if not torch.is_tensor(waveform):
waveform = torch.as_tensor(waveform, dtype=torch.float32)
waveform = waveform.detach().to(device="cpu", dtype=torch.float32)
if waveform.dim() == 1:
waveform = waveform.unsqueeze(0)
elif waveform.dim() == 3 and waveform.shape[0] == 1:
waveform = waveform.squeeze(0)
if waveform.dim() != 2:
raise ValueError(f"audio.cpp waveform must be [channels, samples], got {tuple(waveform.shape)}")
if sample_rate is None or int(sample_rate) <= 0:
raise ValueError("audio.cpp returned an invalid sample rate")
return waveform.contiguous(), int(sample_rate)
def generate_single(
self,
text: str,
voice_ref: Optional[Dict[str, Any]] = None,
seed: int = 0,
enable_audio_cache: bool = True,
character_name: Optional[str] = None,
) -> Tuple[torch.Tensor, int]:
stripped = str(text or "").strip()
if not stripped:
if self._last_sample_rate is None:
raise ValueError("audio.cpp cannot determine a sample rate for empty text")
return torch.zeros(1, 0, dtype=torch.float32), self._last_sample_rate
session = _get_session(self.config)
task = self._resolved_task(session)
advanced = self._advanced_options()
cleanup_path: Optional[str] = None
try:
voice_path, reference_text, audio_component, cleanup_path = self._materialize_reference(voice_ref)
cache_key = self._cache_key(
stripped,
audio_component,
reference_text,
seed,
task,
advanced,
character_name,
session,
)
if enable_audio_cache:
cached = self.audio_cache.get_cached_audio(cache_key)
with _CACHE_SAMPLE_RATES_LOCK:
cached_rate = _CACHE_SAMPLE_RATES.get(cache_key)
if cached is not None and cached_rate is not None:
self._last_sample_rate = cached_rate
return cached[0].clone(), cached_rate
request = self._build_request(stripped, voice_path, reference_text, seed, advanced, task)
waveform, sample_rate = self._normalize_result(session.run(request))
self._last_sample_rate = sample_rate
if enable_audio_cache:
duration = waveform.shape[-1] / sample_rate
self.audio_cache.cache_audio(cache_key, waveform, duration)
with _CACHE_SAMPLE_RATES_LOCK:
_CACHE_SAMPLE_RATES[cache_key] = sample_rate
return waveform, sample_rate
finally:
if cleanup_path:
try:
os.remove(cleanup_path)
except FileNotFoundError:
pass
# Short alias for callers that do not use the older ``EngineAdapter`` suffix.
AudioCppAdapter = AudioCppEngineAdapter
+111
View File
@@ -0,0 +1,111 @@
"""audio.cpp adapter for the Suite's unified Voice Changer node."""
from __future__ import annotations
import json
import os
from typing import Any, Dict, Mapping
import torch
from engines.adapters.audio_cpp_adapter import AudioCppEngineAdapter
from utils.audio.processing import AudioProcessingUtils
def _advanced_options(config: Mapping[str, Any]) -> Dict[str, Any]:
value = config.get("advanced_options", config.get("request_options", {}))
if value in (None, ""):
return {}
if isinstance(value, str):
try:
value = json.loads(value)
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
if not isinstance(value, Mapping):
raise ValueError("audio.cpp advanced options must be a JSON object")
return dict(value)
def _materialize(audio: Mapping[str, Any], label: str) -> str:
waveform = audio.get("waveform")
sample_rate = audio.get("sample_rate")
if not torch.is_tensor(waveform):
raise TypeError(f"audio.cpp {label} must contain a waveform tensor")
if sample_rate is None or int(sample_rate) <= 0:
raise ValueError(f"audio.cpp {label} must contain a positive sample_rate")
return os.path.abspath(
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
)
class AudioCppVoiceConversionAdapter:
"""Convert source audio toward a target reference using an audio.cpp VC task."""
def __init__(self, config: Dict[str, Any]):
self.config = dict(config)
def _session_config(self) -> Dict[str, Any]:
config = dict(self.config)
if str(config.get("connection_mode", "auto")).lower() != "external_server":
config["requested_task"] = "vc"
config["task"] = "vc"
return config
def convert_voice(
self,
source_audio: Dict[str, Any],
target_audio: Dict[str, Any],
refinement_passes: int = 1,
) -> tuple[Dict[str, Any], str]:
from utils.audio_cpp.session import get_audio_cpp_session
config = self._session_config()
family = str(config.get("family", "")).strip()
passes = max(1, int(refinement_passes))
current = source_audio
output_rate = int(source_audio["sample_rate"])
session = get_audio_cpp_session(config)
if str(getattr(session, "task", "vc")) != "vc":
raise ValueError(
f"audio.cpp model '{session.model_id}' is configured for task "
f"'{session.task}', not voice conversion"
)
for pass_index in range(passes):
source_path = _materialize(current, "source audio")
target_path = _materialize(target_audio, "target reference audio")
try:
request = {
"audio": source_path,
"voice_ref": target_path,
"source_audio": source_path,
"target_voice": target_path,
"options": _advanced_options(config),
}
print(
f"🔄 audio.cpp VC: {family or 'external model'} pass "
f"{pass_index + 1}/{passes}..."
)
result = session.run(request)
waveform, output_rate = AudioCppEngineAdapter._normalize_result(result)
current = {"waveform": waveform.unsqueeze(0), "sample_rate": output_rate}
finally:
for path in (source_path, target_path):
try:
os.remove(path)
except FileNotFoundError:
pass
info = (
f"Model family: {family or getattr(session, 'family', 'external')}\n"
f"Model ID: {session.model_id}\n"
f"Task: voice conversion\n"
f"Refinement passes: {passes}\n"
f"Output sample rate: {output_rate} Hz\n"
"Conversion completed successfully"
)
return current, info
__all__ = ["AudioCppVoiceConversionAdapter"]
+12
View File
@@ -195,6 +195,14 @@ except Exception as e:
print(f"❌ Fish Audio S2 Pro Engine failed: {e}")
FISH_AUDIO_S2_ENGINE_AVAILABLE = False
try:
audio_cpp_engine_module = load_node_module("audio_cpp_engine_node", "engines/audio_cpp_engine_node.py")
AudioCppEngineNode = audio_cpp_engine_module.AudioCppEngineNode
AUDIO_CPP_ENGINE_AVAILABLE = True
except Exception as e:
print(f"❌ audio.cpp Engine failed: {e}")
AUDIO_CPP_ENGINE_AVAILABLE = False
try:
omnivoice_engine_module = load_node_module("omnivoice_engine_node", "engines/omnivoice_engine_node.py")
OmniVoiceEngineNode = omnivoice_engine_module.OmniVoiceEngineNode
@@ -702,6 +710,10 @@ if FISH_AUDIO_S2_ENGINE_AVAILABLE:
NODE_CLASS_MAPPINGS["FishAudioS2EngineNode"] = FishAudioS2EngineNode
NODE_DISPLAY_NAME_MAPPINGS["FishAudioS2EngineNode"] = "⚙️ Fish Audio S2 Pro Engine"
if AUDIO_CPP_ENGINE_AVAILABLE:
NODE_CLASS_MAPPINGS["AudioCppEngineNode"] = AudioCppEngineNode
NODE_DISPLAY_NAME_MAPPINGS["AudioCppEngineNode"] = "⚙️ audio.cpp Multi-TTS Engine"
if OMNIVOICE_ENGINE_AVAILABLE:
NODE_CLASS_MAPPINGS["OmniVoiceEngineNode"] = OmniVoiceEngineNode
NODE_DISPLAY_NAME_MAPPINGS["OmniVoiceEngineNode"] = "⚙️ OmniVoice Engine"
+16
View File
@@ -0,0 +1,16 @@
"""audio.cpp processor exports."""
from .audio_cpp_processor import AudioCPPProcessor, AudioCppProcessor
from .audio_cpp_srt_processor import (
AudioCPPSRTProcessor,
AudioCppSRTProcessor,
AudioCppSubtitleProcessor,
)
__all__ = [
"AudioCppProcessor",
"AudioCPPProcessor",
"AudioCppSRTProcessor",
"AudioCPPSRTProcessor",
"AudioCppSubtitleProcessor",
]
+412
View File
@@ -0,0 +1,412 @@
"""Text orchestration for the generic audio.cpp TTS engine."""
from __future__ import annotations
import re
from typing import Any, Dict, List, Mapping, Optional, Tuple, Union
import torch
from utils.audio.chunk_combiner import ChunkCombiner
from utils.text.character_parser import character_parser
from utils.text.pause_processor import PauseTagProcessor
from utils.text.segment_parameters import ParameterValidator, apply_segment_parameters
from utils.text.step_audio_editx_special_tags import get_edit_tags_for_segment
from utils.voice.character_logging import (
format_resolved_character_block,
resolved_character_label,
)
from utils.voice.discovery import get_available_characters, get_character_mapping, voice_discovery
from utils.voice.reference import effective_voice_audio
class AudioCppProcessor:
"""Apply suite text features while accepting the runtime's response sample rate."""
_RUNTIME_KEYS = (
"connection_mode",
"server_url",
"external_server_url",
"binary_path",
"family",
"package_id",
"model_path",
"model_id",
"task",
"backend",
"device",
)
def __init__(self, adapter: Any, engine_config: Optional[Dict[str, Any]] = None):
self.adapter = adapter
self.config = dict(engine_config or {})
self._sample_rate: Optional[int] = None
@property
def sample_rate(self) -> Optional[int]:
return self._sample_rate
def update_config(self, new_config: Optional[Dict[str, Any]]) -> None:
new_value = dict(new_config or {})
old_signature = tuple(self.config.get(key) for key in self._RUNTIME_KEYS)
new_signature = tuple(new_value.get(key) for key in self._RUNTIME_KEYS)
if old_signature != new_signature:
self._sample_rate = None
self.config = new_value
self.adapter.update_config(new_value)
def reset_sample_rate(self) -> None:
"""Begin a top-level generation without retaining an old server rate."""
self._sample_rate = None
@staticmethod
def _check_interrupt() -> None:
try:
import comfy.model_management as model_management
if getattr(model_management, "interrupt_processing", False) is True:
raise InterruptedError("audio.cpp generation interrupted by user")
except ImportError:
return
def _adopt_sample_rate(self, sample_rate: Any) -> int:
try:
value = int(sample_rate)
except (TypeError, ValueError) as exc:
raise ValueError(f"audio.cpp returned invalid sample rate: {sample_rate!r}") from exc
if value <= 0:
raise ValueError(f"audio.cpp returned invalid sample rate: {value}")
if self._sample_rate is None:
self._sample_rate = value
elif self._sample_rate != value:
raise RuntimeError(
"audio.cpp returned inconsistent sample rates in one generation "
f"({self._sample_rate} Hz then {value} Hz)"
)
return value
def _setup_character_parser(self, text: str) -> None:
language = str(self.config.get("language", "auto") or "auto").strip()
fallback = "en" if language.lower() in {"", "auto", "none"} else language.lower()
character_parser.language_resolver.default_language = fallback
character_parser.default_language = fallback
tagged = []
for raw in re.findall(r"\[([^\]]+)\]", text or ""):
name = raw.split("|", 1)[0].strip()
if name and not name.lower().startswith(("pause:", "wait:", "stop:")):
tagged.append(name)
available = {str(item).lower() for item in (get_available_characters() or [])}
for alias, target in voice_discovery.get_character_aliases().items():
available.update((str(alias).lower(), str(target).lower()))
available.update(name.lower() for name in tagged)
available.add("narrator")
character_parser.set_available_characters(sorted(available))
for character, default_language in voice_discovery.get_character_language_defaults().items():
character_parser.set_character_language_default(character, default_language)
character_parser.reset_session_cache()
@staticmethod
def _should_apply_segment_language(segment: Any, base_config: Mapping[str, Any]) -> bool:
language = str(getattr(segment, "language", "") or "").strip()
if not language:
return False
if getattr(segment, "explicit_language", False):
return True
global_language = str(base_config.get("language", "auto") or "auto").strip().lower()
parser_fallback = str(character_parser.default_language or "").strip().lower()
return language.lower() != parser_fallback and language.lower() != global_language
@staticmethod
def _voice_for_character(
character: str,
voice_mapping: Mapping[str, Any],
discovered: Mapping[str, Tuple[Optional[str], Optional[str]]],
) -> Dict[str, Any]:
narrator = voice_mapping.get("narrator", {})
voice = dict(narrator) if isinstance(narrator, Mapping) else {"audio": narrator}
if character != "narrator" and character in voice_mapping:
selected = voice_mapping[character]
return dict(selected) if isinstance(selected, Mapping) else {"audio": selected}
if character != "narrator":
audio_path, reference_text = discovered.get(character, (None, None))
if audio_path:
return {"audio_path": audio_path, "reference_text": reference_text or ""}
return voice
@staticmethod
def _chunks(text: str, enabled: bool, max_chars: int) -> List[str]:
if not enabled:
return [text]
from utils.text.chunking import ImprovedChatterBoxChunker
limit = ImprovedChatterBoxChunker.validate_chunking_params(max_chars)
return ImprovedChatterBoxChunker.split_into_chunks(text, max_chars=limit)
@staticmethod
def _voice_log_note(voice_ref: Mapping[str, Any]) -> str:
if not isinstance(voice_ref, Mapping) or effective_voice_audio(voice_ref) is None:
return " [no voice reference - model default]"
reference_text = str(voice_ref.get("reference_text") or "").strip()
if reference_text:
return f" [ref text: {len(reference_text)} chars]"
return ""
@staticmethod
def _format_parameter_log(
parameters: Mapping[str, Any], current_config: Mapping[str, Any], current_seed: int
) -> str:
if not parameters:
return ""
parts = []
for key in parameters:
if key == "seed":
value = current_seed
else:
value = current_config.get(key, parameters.get(key))
if value is not None and value != "":
parts.append(f"{key}={value}")
return ", ".join(parts)
def _log_generation_text(
self,
character: str,
text: str,
voice_ref: Mapping[str, Any],
language: str,
family: str,
chunk_count: int,
parameter_log: str,
) -> None:
display_name = resolved_character_label(character, voice_ref)
voice_note = self._voice_log_note(voice_ref)
print(
f"🎭 Audio.cpp ({family}) - Generating for '{display_name}' "
f"(Language: {language}){voice_note}:"
)
if parameter_log:
print(f"🎛️ Audio.cpp params: {parameter_log}")
print(format_resolved_character_block(character, text, voice_ref))
if chunk_count > 1:
print(
f"📝 Chunking {display_name}'s text into {chunk_count} chunks "
f"(Language: {language}){voice_note}"
)
def get_character_order(self, text: str) -> List[str]:
self._setup_character_parser(text)
seen: List[str] = []
for segment in character_parser.parse_text_segments(text, engine_type="audio_cpp"):
character = segment.character or "narrator"
if character not in seen:
seen.append(character)
return seen
def process_text(
self,
text: str,
voice_mapping: Optional[Dict[str, Any]],
seed: int,
enable_chunking: bool = True,
max_chars_per_chunk: int = 400,
chunk_combination_method: str = "auto",
silence_between_chunks_ms: int = 100,
enable_audio_cache: bool = True,
apply_edit_postprocessing: bool = True,
show_text_logging: bool = True,
reset_sample_rate: bool = True,
**_: Any,
) -> List[Dict[str, Any]]:
del chunk_combination_method, silence_between_chunks_ms
if reset_sample_rate:
self.reset_sample_rate()
self._check_interrupt()
voice_mapping = dict(voice_mapping or {})
self._setup_character_parser(text)
base_config = self.config.copy()
segments = character_parser.parse_text_segments(text, engine_type="audio_cpp")
if not segments and str(text or "").strip():
segments = character_parser.parse_text_segments(
f"[narrator]{text}", engine_type="audio_cpp"
)
characters = list({segment.character for segment in segments if segment.character})
# GLM-TTS requires the transcript paired with its reference voice.
# Other pinned families accept audio-only discovery and still receive a
# transcript whenever one exists beside the character audio file.
try:
from utils.audio_cpp.capabilities import get_capability
transcript_requirement = get_capability(
str(base_config.get("family", ""))
)["reference_transcript"]
except (ImportError, KeyError, ValueError):
transcript_requirement = "none"
discovery_type = (
"audio_and_text" if transcript_requirement == "required" else "audio_only"
)
discovered = get_character_mapping(characters, engine_type=discovery_type)
configured_speakers = list(base_config.get("speaker_references") or [])
ordered_characters = []
for segment in segments:
name = segment.character or "narrator"
if name not in ordered_characters:
ordered_characters.append(name)
for index, reference in enumerate(configured_speakers, start=1):
if index < len(ordered_characters):
selected = reference if isinstance(reference, Mapping) else {"audio": reference}
voice_mapping[ordered_characters[index]] = dict(selected)
records: List[Dict[str, Any]] = []
for segment in segments:
self._check_interrupt()
segment_text = str(segment.text or "").strip()
if not segment_text:
continue
character = segment.character or "narrator"
parameters = dict(segment.parameters or {})
filtered_parameters: Dict[str, Any] = {}
current_config = base_config
current_seed = int(seed)
if parameters:
filtered_parameters = ParameterValidator.filter_parameters_for_engine(
parameters, "audio_cpp"
)
current_config = apply_segment_parameters(base_config, parameters, "audio_cpp")
current_seed = int(current_config.get("seed", seed))
if self._should_apply_segment_language(segment, base_config):
current_config = current_config.copy()
current_config["language"] = segment.language
self.adapter.update_config(current_config)
voice_ref = self._voice_for_character(character, voice_mapping, discovered)
try:
from utils.audio_cpp.capabilities import CapabilityError, validate_voice_reference
except ImportError:
validate_voice_reference = None
if validate_voice_reference is not None:
try:
validate_voice_reference(
str(base_config.get("family", "")), voice_ref, character
)
except CapabilityError:
# Preserve lightweight processor use before a concrete family
# has been selected, while enforcing every known family.
pass
def generate_fragment(content: str, edit_tags: List[Any]) -> None:
chunks = self._chunks(content, enable_chunking, max_chars_per_chunk)
if show_text_logging:
language = str(current_config.get("language", "auto") or "auto")
family = str(current_config.get("family", "unknown") or "unknown")
self._log_generation_text(
character,
content,
voice_ref,
language,
family,
len(chunks),
self._format_parameter_log(
filtered_parameters, current_config, current_seed
),
)
for chunk_index, chunk in enumerate(chunks):
self._check_interrupt()
waveform, response_rate = self.adapter.generate_single(
text=chunk,
voice_ref=voice_ref,
seed=current_seed + chunk_index,
enable_audio_cache=enable_audio_cache,
character_name=character,
)
sample_rate = self._adopt_sample_rate(response_rate)
waveform = waveform.detach().to(device="cpu", dtype=torch.float32)
if waveform.dim() == 1:
waveform = waveform.unsqueeze(0)
if waveform.dim() != 2:
raise ValueError(
f"audio.cpp waveform must be [channels, samples], got {tuple(waveform.shape)}"
)
records.append(
{
"waveform": waveform,
"sample_rate": sample_rate,
"text": chunk,
"edit_tags": edit_tags if chunk_index == 0 else [],
}
)
if PauseTagProcessor.has_pause_tags(segment_text):
pause_parts, _ = PauseTagProcessor.parse_pause_tags(segment_text)
for part_type, content in pause_parts:
if part_type == "text":
clean_text, edit_tags = get_edit_tags_for_segment(str(content))
if clean_text.strip():
generate_fragment(clean_text.strip(), edit_tags)
else:
records.append(
{
"pause_duration": float(content),
"text": f"[pause:{content}s]",
"edit_tags": [],
}
)
else:
clean_text, edit_tags = get_edit_tags_for_segment(segment_text)
if clean_text.strip():
generate_fragment(clean_text.strip(), edit_tags)
self.adapter.update_config(base_config)
if any("pause_duration" in record for record in records):
if self._sample_rate is None:
raise ValueError("audio.cpp cannot render pauses before any response sample rate is known")
for record in records:
if "pause_duration" not in record:
continue
record["waveform"] = PauseTagProcessor.create_silence_segment(
record.pop("pause_duration"), self._sample_rate, torch.device("cpu"), torch.float32
)
record["sample_rate"] = self._sample_rate
if apply_edit_postprocessing and records and any(record.get("edit_tags") for record in records):
from utils.audio.edit_post_processor import process_segments as apply_edits
records = apply_edits(records, engine_config=base_config)
for record in records:
self._adopt_sample_rate(record.get("sample_rate"))
return records
def combine_audio_segments(
self,
segments: List[Dict[str, Any]],
method: str = "auto",
silence_ms: int = 100,
original_text: str = "",
return_info: bool = False,
) -> Union[torch.Tensor, Tuple[torch.Tensor, Dict[str, Any]]]:
if not segments:
empty = torch.zeros(0, dtype=torch.float32)
return (empty, {}) if return_info else empty
rates = {self._adopt_sample_rate(segment.get("sample_rate")) for segment in segments}
if len(rates) != 1:
raise RuntimeError(f"audio.cpp segments use inconsistent sample rates: {sorted(rates)}")
sample_rate = rates.pop()
waveforms = [segment["waveform"] for segment in segments]
text_chunks = [str(segment.get("text", "")) for segment in segments]
result = ChunkCombiner.combine_chunks(
audio_segments=waveforms,
method=method,
silence_ms=int(silence_ms),
crossfade_duration=0.1,
sample_rate=sample_rate,
text_length=len(" ".join(text_chunks)),
original_text=original_text,
text_chunks=text_chunks,
return_info=return_info,
)
return result
# Compatibility with integration code that uses an all-caps acronym.
AudioCPPProcessor = AudioCppProcessor
+238
View File
@@ -0,0 +1,238 @@
"""SRT timing orchestration for audio.cpp with a response-defined sample rate."""
from __future__ import annotations
import importlib.util
import os
from typing import Any, Dict, List, Optional, Tuple
import torch
from utils.system.import_manager import import_manager
from utils.timing.assembly import AudioAssemblyEngine
from utils.timing.engine import TimingEngine
from utils.timing.overlap_detection import SRTOverlapHandler
from utils.timing.reporting import SRTReportGenerator
def _processor_class():
"""Load by path because this project also has a top-level ``nodes.py`` module."""
path = os.path.join(os.path.dirname(__file__), "audio_cpp_processor.py")
spec = importlib.util.spec_from_file_location("audio_cpp_processor_module", path)
if spec is None or spec.loader is None:
raise ImportError(f"Cannot load audio.cpp processor from {path}")
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module.AudioCppProcessor
def _adapter_class():
"""Load directly so unrelated optional adapters are not imported eagerly."""
path = os.path.abspath(
os.path.join(os.path.dirname(__file__), "..", "..", "engines", "adapters", "audio_cpp_adapter.py")
)
spec = importlib.util.spec_from_file_location("audio_cpp_adapter_module", path)
if spec is None or spec.loader is None:
raise ImportError(f"Cannot load audio.cpp adapter from {path}")
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module.AudioCppEngineAdapter
class AudioCppSRTProcessor:
"""Generate one subtitle cue at a time and assemble it on the SRT timeline."""
def __init__(self, node_instance: Any, config: Optional[Dict[str, Any]] = None):
self.node_instance = node_instance
self.config = dict(config or {})
self.adapter = _adapter_class()(self.config)
self._processor = _processor_class()(self.adapter, self.config)
success, modules, message = import_manager.import_srt_modules()
if not success or modules.get("SRTParser") is None:
raise ImportError(f"audio.cpp SRT unavailable: {message}")
self.SRTParser = modules["SRTParser"]
@property
def processor(self) -> Any:
return self._processor
@property
def sample_rate(self) -> Optional[int]:
return self.processor.sample_rate
def update_config(self, config: Optional[Dict[str, Any]]) -> None:
self.config = dict(config or {})
self.processor.update_config(self.config)
@staticmethod
def _check_interrupt(index: Optional[int] = None, total: Optional[int] = None) -> None:
try:
import comfy.model_management as model_management
if getattr(model_management, "interrupt_processing", False) is True:
location = f" at subtitle {index + 1}/{total}" if index is not None else ""
raise InterruptedError(f"audio.cpp SRT generation interrupted{location}")
except ImportError:
return
@staticmethod
def _adjustment(index: int, subtitle: Any, audio: torch.Tensor, sample_rate: int) -> Dict[str, Any]:
natural = audio.shape[-1] / sample_rate
target = float(subtitle.duration)
ratio = target / natural if natural > 0 else 1.0
return {
"index": index,
"segment_index": index,
"sequence": subtitle.sequence,
"natural_duration": natural,
"target_start": subtitle.start_time,
"target_end": subtitle.end_time,
"target_duration": target,
"start_time": subtitle.start_time,
"end_time": subtitle.end_time,
"stretch_factor": ratio,
"needs_stretching": abs(ratio - 1.0) > 0.05,
"stretch_type": "compress" if ratio < 1 else "expand" if ratio > 1 else "none",
"adjustment": natural - target,
"adjusted_start": subtitle.start_time,
"adjusted_end": subtitle.end_time,
"adjusted_duration": natural,
}
def process_srt_content(
self,
srt_content: str,
voice_mapping: Optional[Dict[str, Any]],
seed: int,
timing_mode: str,
timing_params: Optional[Dict[str, Any]],
enable_audio_cache: bool = True,
) -> Tuple[Dict[str, Any], str, str, str]:
self._check_interrupt()
subtitles = self.SRTParser().parse_srt_content(srt_content, allow_overlaps=True)
if not subtitles:
raise ValueError("audio.cpp SRT input contains no subtitles")
has_overlaps = SRTOverlapHandler.detect_overlaps(subtitles)
active_mode, switched = SRTOverlapHandler.handle_smart_natural_fallback(
timing_mode, has_overlaps, "audio.cpp SRT"
)
self.processor.reset_sample_rate()
audio_segments: List[Optional[torch.Tensor]] = []
for index, subtitle in enumerate(subtitles):
self._check_interrupt(index, len(subtitles))
text = str(subtitle.text or "").strip()
if not text:
audio_segments.append(None)
continue
records = self.processor.process_text(
text=text,
voice_mapping=voice_mapping or {},
seed=int(seed) + index,
enable_chunking=False,
enable_audio_cache=enable_audio_cache,
apply_edit_postprocessing=True,
show_text_logging=True,
reset_sample_rate=False,
)
if not records:
raise RuntimeError(f"audio.cpp produced no audio for subtitle {index + 1}")
audio = self.processor.combine_audio_segments(
records, method="auto", silence_ms=0, original_text=text
)
if audio.dim() == 1:
audio = audio.unsqueeze(0)
elif audio.dim() == 3 and audio.shape[0] == 1:
audio = audio.squeeze(0)
audio_segments.append(audio.detach().to(device="cpu", dtype=torch.float32))
sample_rate = self.processor.sample_rate
if sample_rate is None:
raise ValueError("audio.cpp could not determine a sample rate from the SRT content")
completed_segments: List[torch.Tensor] = []
for subtitle, audio in zip(subtitles, audio_segments):
if audio is None:
audio = torch.zeros(1, int(float(subtitle.duration) * sample_rate), dtype=torch.float32)
completed_segments.append(audio)
adjustments = [
self._adjustment(index, subtitle, completed_segments[index], sample_rate)
for index, subtitle in enumerate(subtitles)
]
self._check_interrupt()
final_audio, replacement, stretch_method = self._assemble(
completed_segments, subtitles, active_mode, dict(timing_params or {}), sample_rate
)
if replacement is not None:
adjustments = replacement
reporter = SRTReportGenerator()
report = reporter.generate_timing_report(
subtitles,
adjustments,
active_mode,
has_overlaps,
switched,
timing_mode if switched else None,
stretch_method,
)
adjusted_srt = reporter.generate_adjusted_srt_string(subtitles, adjustments, active_mode)
if final_audio.dim() == 1:
final_audio = final_audio.unsqueeze(0).unsqueeze(0)
elif final_audio.dim() == 2:
final_audio = final_audio.unsqueeze(0)
duration = final_audio.shape[-1] / sample_rate
mode_info = f"{active_mode} (switched from {timing_mode})" if switched else active_mode
info = (
f"Generated {duration:.1f}s audio.cpp SRT audio from {len(subtitles)} subtitles "
f"using {mode_info} mode at {sample_rate} Hz"
)
return {"waveform": final_audio, "sample_rate": sample_rate}, info, report, adjusted_srt
@staticmethod
def _assemble(
audio_segments: List[torch.Tensor],
subtitles: List[Any],
mode: str,
params: Dict[str, Any],
sample_rate: int,
):
fade = params.get("fade_for_StretchToFit", 0.01)
if mode == "stretch_to_fit":
from engines.chatterbox.audio_timing import TimedAudioAssembler
assembler = TimedAudioAssembler(sample_rate)
audio, method = assembler.assemble_timed_audio(
audio_segments,
[(item.start_time, item.end_time) for item in subtitles],
fade_duration=fade,
)
return audio, None, method
assembler = AudioAssemblyEngine(sample_rate)
if mode == "pad_with_silence":
audio = assembler.assemble_with_overlaps(audio_segments, subtitles, torch.device("cpu"))
return audio, None, None
timing = TimingEngine(sample_rate)
if mode == "concatenate":
replacements = timing.calculate_concatenation_adjustments(audio_segments, subtitles)
audio = assembler.assemble_concatenation(audio_segments, fade)
return audio, replacements, None
replacements, processed = timing.calculate_smart_timing_adjustments(
audio_segments,
subtitles,
params.get("timing_tolerance", 2.0),
params.get("max_stretch_ratio", 1.0),
params.get("min_stretch_ratio", 0.5),
torch.device("cpu"),
)
audio = assembler.assemble_smart_natural(
audio_segments, processed, replacements, subtitles, torch.device("cpu")
)
return audio, replacements, None
AudioCppSubtitleProcessor = AudioCppSRTProcessor
AudioCPPSRTProcessor = AudioCppSRTProcessor
+444
View File
@@ -0,0 +1,444 @@
"""ComfyUI configuration node for the generic audio.cpp backend."""
from __future__ import annotations
import glob
import json
import os
from typing import Any, Dict, List, Mapping, Optional
from urllib.parse import urlparse
class AnyType(str):
def __ne__(self, __value: object) -> bool:
return False
any_type = AnyType("*")
def _catalog_module():
try:
from utils.audio_cpp import catalog
return catalog
except ImportError:
return None
def _fallback_specs() -> List[Dict[str, Any]]:
root = os.path.abspath(
os.path.join(os.path.dirname(__file__), "..", "..", "utils", "audio_cpp", "model_specs")
)
specs = []
for path in glob.glob(os.path.join(root, "*.json")):
try:
with open(path, "r", encoding="utf-8") as handle:
value = json.load(handle)
if isinstance(value, dict) and value.get("family"):
specs.append(value)
except (OSError, json.JSONDecodeError):
continue
return specs
def _family_choices() -> List[str]:
catalog = _catalog_module()
if catalog is not None and callable(getattr(catalog, "family_choices", None)):
choices = list(catalog.family_choices())
else:
choices = [spec["family"] for spec in _fallback_specs()]
choices = sorted({str(choice) for choice in choices if str(choice).strip()})
return choices or ["qwen3_tts"]
def _package_choices() -> List[str]:
catalog = _catalog_module()
if catalog is not None and callable(getattr(catalog, "package_choices", None)):
choices = list(catalog.package_choices())
else:
choices = [
package.get("id")
for spec in _fallback_specs()
for package in spec.get("packages", [])
if isinstance(package, dict)
]
return ["auto"] + sorted({str(choice) for choice in choices if choice})
def _recommended_package(family: str) -> str:
catalog = _catalog_module()
if catalog is not None and callable(getattr(catalog, "recommended_package", None)):
value = catalog.recommended_package(family)
if value:
return str(value)
for spec in _fallback_specs():
if spec.get("family") != family:
continue
recommended = (spec.get("ui") or {}).get("recommended_package")
if recommended:
return str(recommended)
for package in spec.get("packages", []):
if package.get("default"):
return str(package["id"])
return "auto"
def _resolve_task(family: str, package_id: str, requested: str) -> str:
requested = str(requested or "auto").lower()
if requested in {"tts", "clon", "vdes", "vc", "s2s", "svc", "asr", "diar"}:
return requested
catalog = _catalog_module()
if catalog is not None and callable(getattr(catalog, "resolve_task", None)):
return str(catalog.resolve_task(family, package_id, requested="auto")).lower()
package_lower = package_id.lower()
if "voicedesign" in package_lower or "voice_design" in package_lower:
return "vdes"
if family in {"chatterbox", "confucius4_tts"}:
return "clon"
return "tts"
def _validate_package(family: str, package_id: str) -> None:
catalog = _catalog_module()
getter = getattr(catalog, "get_package", None) if catalog is not None else None
if not callable(getter) or package_id == "auto":
return
value = getter(package_id)
if value is None:
raise ValueError(f"Unknown audio.cpp package: {package_id}")
package_family = value.get("family") if isinstance(value, Mapping) else getattr(value, "family", None)
if package_family and str(package_family) != family:
raise ValueError(f"audio.cpp package '{package_id}' does not belong to family '{family}'")
class AudioCppEngineNode:
"""Describe either a managed audio.cpp runtime or an existing installation."""
@classmethod
def NAME(cls):
return "⚙️ audio.cpp Multi-TTS Engine"
@classmethod
def INPUT_TYPES(cls):
families = _family_choices()
default_family = "qwen3_tts" if "qwen3_tts" in families else families[0]
packages = _package_choices()
return {
"required": {
"connection_mode": (
["auto", "external_server", "existing_binary", "managed"],
{
"default": "auto",
"tooltip": "Auto prefers a supplied server or binary, then the suite-managed runtime.",
},
),
"family": (
families,
{
"default": default_family,
"tooltip": "audio.cpp model family. The package list and capability panel update to match this selection.",
},
),
"package_id": (
packages,
{
"default": "auto",
"tooltip": "Auto selects the pinned recommended package for the chosen family.",
},
),
"task": (
["auto", "tts", "clon", "vdes", "vc", "s2s", "svc", "asr", "diar"],
{
"default": "auto",
"tooltip": "Runtime task. Auto lets the connected unified node use the family's normal task; choose an explicit task only for advanced routing or external-server matching.",
},
),
"backend": (
["auto", "cuda", "cpu", "vulkan", "metal", "hip"],
{
"default": "auto",
"tooltip": "Native audio.cpp compute backend. Auto selects an installed CUDA runtime when available, otherwise CPU.",
},
),
"device": (
"INT",
{
"default": 0,
"min": 0,
"max": 31,
"tooltip": "Zero-based native device index. Keep 0 unless using another GPU/device.",
},
),
"threads": (
"INT",
{
"default": 4,
"min": 1,
"max": 128,
"tooltip": "Native backend/OpenMP workers. Four matches the audio.cpp CLI default; tune for your CPU.",
},
),
"language": (
"STRING",
{
"default": "auto",
"tooltip": "Language code passed to audio.cpp. Auto lets the selected model infer or use its default language.",
},
),
},
"optional": {
"server_url": (
"STRING",
{
"default": "",
"tooltip": "Required only for external_server mode, for example http://127.0.0.1:8080.",
},
),
"binary_path": (
"STRING",
{
"default": "",
"tooltip": "Optional path to an existing audiocpp_server executable. Leave blank to use the Suite-managed runtime.",
},
),
"model_path": (
"STRING",
{
"default": "",
"tooltip": "Optional existing audio.cpp model/package directory. Leave blank for discovery or managed download.",
},
),
"model_id": (
"STRING",
{
"default": "",
"tooltip": "Server model identifier. Usually leave blank; required when an external server exposes multiple models.",
},
),
"voice_id": (
"STRING",
{
"default": "",
"tooltip": "Optional built-in voice/preset ID for families such as Supertonic. Reference audio takes precedence when supported.",
},
),
"instruct": (
"STRING",
{
"default": "",
"multiline": True,
"tooltip": "Optional natural-language voice design or style instruction. Used only by families/tasks that support instructions.",
},
),
"speaker2": (any_type, {"tooltip": "Optional ordered character/Speaker 2 reference."}),
"temperature": ("FLOAT", {"default": -1.0, "min": -1.0, "max": 5.0, "step": 0.05, "tooltip": "Sampling temperature. -1 uses the selected model/package default."}),
"top_p": ("FLOAT", {"default": -1.0, "min": -1.0, "max": 1.0, "step": 0.01, "tooltip": "Nucleus sampling threshold. -1 uses the model default."}),
"top_k": ("INT", {"default": -1, "min": -1, "max": 1000, "tooltip": "Top-k sampling limit. -1 uses the model default."}),
"repetition_penalty": (
"FLOAT",
{"default": -1.0, "min": -1.0, "max": 5.0, "step": 0.05, "tooltip": "Token repetition penalty. -1 uses the model default."},
),
"max_tokens": ("INT", {"default": 0, "min": 0, "max": 131072, "tooltip": "Maximum generated tokens. 0 lets the model choose its normal limit."}),
"max_steps": ("INT", {"default": 0, "min": 0, "max": 4096, "tooltip": "Maximum generation/decoder steps where supported. 0 uses the model default."}),
"num_inference_steps": ("INT", {"default": 0, "min": 0, "max": 1000, "tooltip": "Flow/diffusion inference steps where supported. 0 uses the model default."}),
"guidance_scale": (
"FLOAT",
{"default": -1.0, "min": -1.0, "max": 100.0, "step": 0.05, "tooltip": "Classifier-free guidance scale where supported. -1 uses the model default."},
),
"advanced_json": (
"STRING",
{
"default": "{}",
"multiline": True,
"tooltip": "Model-specific audio.cpp request options as a JSON object.",
},
),
"auto_download_runtime": (
"BOOLEAN",
{
"default": True,
"tooltip": "Automatically install the pinned audio.cpp runtime into Suite-managed storage when no usable runtime is found. Existing external binaries are never copied.",
},
),
"auto_download_model": (
"BOOLEAN",
{
"default": True,
"tooltip": "Automatically download the selected audio.cpp package into models/TTS/audio.cpp/models when it is not already available. Downloads use direct files, not the Hugging Face cache.",
},
),
"show_server_console": (
"BOOLEAN",
{
"default": False,
"tooltip": "Debug only: launch a visible console for a Suite-owned audio.cpp server.",
},
),
},
}
RETURN_TYPES = ("TTS_ENGINE",)
RETURN_NAMES = ("TTS_engine",)
FUNCTION = "create_engine_config"
CATEGORY = "TTS Audio Suite/⚙️ Engines"
def create_engine_config(
self,
connection_mode: str,
family: str,
package_id: str,
task: str,
backend: str,
device: int,
threads: int,
language: str,
server_url: str = "",
binary_path: str = "",
model_path: str = "",
model_id: str = "",
voice_id: str = "",
instruct: str = "",
temperature: float = -1.0,
top_p: float = -1.0,
top_k: int = -1,
repetition_penalty: float = -1.0,
max_tokens: int = 0,
max_steps: int = 0,
num_inference_steps: int = 0,
guidance_scale: float = -1.0,
advanced_json: str = "{}",
auto_download_runtime: bool = True,
auto_download_model: bool = True,
show_server_console: bool = False,
speaker_mode: str = "Custom Character Switching",
speaker2: Any = None,
**kwargs: Any,
) -> tuple:
mode = str(connection_mode).strip().lower()
if mode not in {"auto", "external_server", "existing_binary", "managed"}:
raise ValueError(f"Unsupported audio.cpp connection mode: {connection_mode}")
family = str(family).strip()
package_id = str(package_id or "auto").strip()
if not family:
raise ValueError("audio.cpp family is required")
url = str(server_url or "").strip().rstrip("/")
binary = os.path.abspath(os.path.expanduser(binary_path)) if binary_path.strip() else ""
model = os.path.abspath(os.path.expanduser(model_path)) if model_path.strip() else ""
if mode == "external_server":
parsed = urlparse(url)
if parsed.scheme not in {"http", "https"} or not parsed.netloc:
raise ValueError("audio.cpp external_server mode requires a valid HTTP(S) server_url")
if mode == "existing_binary":
if not binary:
raise ValueError("audio.cpp existing_binary mode requires binary_path")
if not os.path.isfile(binary):
raise FileNotFoundError(f"audio.cpp binary not found: {binary}")
try:
advanced = json.loads(advanced_json or "{}")
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
if not isinstance(advanced, Mapping):
raise ValueError("audio.cpp advanced JSON must contain an object")
uses_existing_server = mode == "external_server"
if package_id == "auto" and not uses_existing_server:
package_id = _recommended_package(family)
if not uses_existing_server:
_validate_package(family, package_id)
resolved_task = _resolve_task(family, package_id, task)
else:
# The loaded model reported by /v1/models owns this decision.
resolved_task = str(task or "auto").lower()
config: Dict[str, Any] = {
"engine_type": "audio_cpp",
"connection_mode": mode,
"family": family,
"package_id": package_id,
"requested_task": str(task or "auto").lower(),
"task": resolved_task,
"backend": str(backend).lower(),
"device": int(device),
"threads": int(threads),
"language": str(language or "auto"),
"server_url": url,
"external_server_url": url,
"binary_path": binary,
"model_path": model,
"model_id": str(model_id or "").strip(),
"voice_id": str(voice_id or "").strip(),
"instruct": str(instruct or "").strip(),
"advanced_options": dict(advanced),
"auto_download_runtime": bool(auto_download_runtime),
"auto_download_model": bool(auto_download_model),
"show_server_console": bool(show_server_console),
"multi_speaker_mode": str(speaker_mode),
}
speakers = [speaker2] if speaker2 is not None else []
dynamic_speakers = []
for key, value in kwargs.items():
if key.startswith("speaker") and key[7:].isdigit() and value is not None:
dynamic_speakers.append((int(key[7:]), value))
speakers.extend(value for _, value in sorted(dynamic_speakers))
config["speaker_references"] = speakers
try:
from utils.audio_cpp.capabilities import get_capability
capability = get_capability(family)
maximum = int(capability["native_multi_speaker"]["max_speakers"])
if len(speakers) > max(0, maximum - 1):
raise ValueError(f"audio.cpp {family} supports at most {maximum} speakers")
if speaker_mode == "Native Multi-Speaker" and capability["native_multi_speaker"]["suite_status"] != "supported":
raise ValueError(
f"audio.cpp {family} native multi-speaker mode is not integrated; "
"use Custom Character Switching"
)
except ImportError:
pass
optional_values = {
"temperature": float(temperature),
"top_p": float(top_p),
"top_k": int(top_k),
"repetition_penalty": float(repetition_penalty),
"guidance_scale": float(guidance_scale),
}
for key, value in optional_values.items():
if value >= 0:
config[key] = value
for key, value in {
"max_tokens": int(max_tokens),
"max_steps": int(max_steps),
"num_inference_steps": int(num_inference_steps),
}.items():
if value > 0:
config[key] = value
try:
from utils.audio_cpp.capabilities import get_capability as load_capability
family_capability = load_capability(family)
suite_tasks = set(family_capability.get("suite_tasks", []))
except (ImportError, KeyError, ValueError):
suite_tasks = {"tts"}
capabilities = []
if "tts" in suite_tasks:
capabilities.append("tts")
if "asr" in suite_tasks:
capabilities.append("asr")
if "voice_conversion" in suite_tasks:
capabilities.append("voice_conversion")
if "diarization" in suite_tasks:
capabilities.append("diarization")
catalog_module = _catalog_module()
family_record = catalog_module.get_family(family) if catalog_module is not None else None
if resolved_task == "vdes" or "vdes" in getattr(family_record, "runtime_tasks", ()):
capabilities.append("voice_design")
return ({"engine_type": "audio_cpp", "config": config, "capabilities": capabilities},)
NODE_CLASS_MAPPINGS = {"AudioCppEngineNode": AudioCppEngineNode}
NODE_DISPLAY_NAME_MAPPINGS = {"AudioCppEngineNode": "⚙️ audio.cpp Multi-TTS Engine"}
+10 -3
View File
@@ -48,7 +48,7 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
return {
"required": {
"engine": ("TTS_ENGINE", {
"tooltip": "ASR-capable engine configuration (for example Qwen3-TTS Engine or Granite ASR Engine). This node auto-routes to the correct ASR adapter based on the engine type."
"tooltip": "ASR-capable engine configuration. Supports Qwen3-TTS ASR, Granite ASR, and audio.cpp families whose capability panel shows ASR. The unified node routes to the correct adapter and preserves available timing/speaker data."
}),
"audio": (any_typ, {
"tooltip": "Audio to transcribe. Accepts AUDIO, Character Voices output, or VideoHelper audio."
@@ -96,7 +96,7 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
}),
"timestamps": (["none", "word"], {
"default": "none",
"tooltip": "Timing detail for the ASR timing output:\n• none: Text only, no reusable timed words/segments\n• word: Word-level timings for timestamp-capable ASR paths\n\nUse word timings if you plan to feed this into the Text to SRT Builder.\n\nGranite note: word timestamps are native on the plus model variant when diarization is off. Other Granite timestamp paths use the separate Qwen forced aligner."
"tooltip": "Timing detail for the ASR timing output:\n• none: Text only, except native speaker turns may still carry segment timing\n• word: Request or preserve word timings when the selected ASR family supports them\n\nUse word timings for Text to SRT Builder.\n\nGranite: the plus model has native timestamps; other variants use the Qwen forced aligner.\naudio.cpp: native words/segments are preserved. Qwen3-ASR specifically needs its optional forced-aligner model for requested word timings and will otherwise continue with text only."
}),
"chunk_size": ("INT", {
"default": 30, "min": 0, "max": 600, "step": 1,
@@ -112,7 +112,7 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
}),
"diarization": ("BOOLEAN", {
"default": False,
"tooltip": "Speaker Diarization (Speaker Attribution):\n• True: Attribute speech to speakers if supported (for example [Speaker 1] hello)\n• False: Plain transcription without speaker turns\n\nGranite note: Native speaker attribution is supported on the 'plus' model variant. If combined with word-level timestamps, the system automatically uses the Qwen forced aligner to time-align the speakers' words."
"tooltip": "Speaker attribution:\n• True: Preserve speaker turns when the selected ASR engine returns them\n• False: Return plain transcription/timing\n\nGranite 4.1 plus and audio.cpp VibeVoice-ASR provide native speaker attribution. Other audio.cpp ASR families return a warning instead of inventing speaker labels."
}),
}
}
@@ -171,6 +171,13 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
engine_cfg = engine.get("config", engine)
cache_data = {
"engine_type": engine.get("engine_type"),
"family": engine_cfg.get("family"),
"package_id": engine_cfg.get("package_id"),
"model_id": engine_cfg.get("model_id"),
"model_path": engine_cfg.get("model_path"),
"connection_mode": engine_cfg.get("connection_mode"),
"server_url": engine_cfg.get("server_url"),
"advanced_options": str(engine_cfg.get("advanced_options", {})),
"model_name": engine_cfg.get("model_name"),
"model_size": engine_cfg.get("model_size"),
"device": engine_cfg.get("device"),
+72
View File
@@ -254,6 +254,16 @@ Hello! This is unified SRT TTS with character switching.
stable_params['dtype'] = config.get('dtype', 'auto')
stable_params['attention'] = config.get('attention', 'auto')
if engine_type == "audio_cpp":
for key in (
'connection_mode', 'server_url', 'server_model_id', 'model_id',
'binary_path', 'model_path', 'model_roots', 'family',
'package_id', 'task', 'backend', 'device', 'device_index',
'threads', 'model_spec_override', 'load_options',
'session_options', 'show_server_console',
):
stable_params[key] = config.get(key)
# IndexTTS 2.0 and 2.5 are distinct checkpoints/backends. Every
# load-time option must participate in the processor cache key or
# changing the engine node can silently keep the old adapter alive.
@@ -995,6 +1005,38 @@ Hello! This is unified SRT TTS with character switching.
}
return engine_instance
elif engine_type == "audio_cpp":
processor_path = os.path.join(nodes_dir, "audio_cpp", "audio_cpp_srt_processor.py")
processor_spec = importlib.util.spec_from_file_location(
"audio_cpp_srt_processor_module", processor_path
)
if processor_spec is None or processor_spec.loader is None:
raise ImportError(f"Cannot load audio.cpp SRT processor from {processor_path}")
processor_module = importlib.util.module_from_spec(processor_spec)
processor_spec.loader.exec_module(processor_module)
AudioCppSRTProcessor = processor_module.AudioCppSRTProcessor
class AudioCppSRTWrapper:
def __init__(self, cfg):
self.config = cfg.copy()
self.processor = AudioCppSRTProcessor(self, self.config)
def update_config(self, new_config):
self.config = new_config.copy()
self.processor.update_config(self.config)
def check_interrupt(self):
if model_management.interrupt_processing:
raise InterruptedError("audio.cpp SRT processing interrupted by user")
engine_instance = AudioCppSRTWrapper(config)
import time
self._cached_engine_instances[cache_key] = {
'instance': engine_instance,
'timestamp': time.time(),
}
return engine_instance
else:
raise ValueError(f"Unknown engine type: {engine_type}")
@@ -1146,6 +1188,11 @@ Hello! This is unified SRT TTS with character switching.
if not engine_type:
raise ValueError("TTS engine missing engine_type")
capabilities = TTS_engine.get("capabilities", [])
if capabilities and "tts" not in capabilities:
raise ValueError(
f"Engine '{engine_type}' does not support TTS/SRT. Connect it to its compatible unified node."
)
if config.get("model_role") == "voice_design":
selected_model = config.get("model_variant") or config.get("model_name") or "selected model"
@@ -1660,6 +1707,29 @@ Hello! This is unified SRT TTS with character switching.
timing_params=timing_params
)
elif engine_type == "audio_cpp":
voice_mapping = {}
if audio_tensor is not None or audio_path:
voice_mapping['narrator'] = {
'audio': audio_tensor,
'audio_path': audio_path,
'reference_text': reference_text or '',
}
timing_params = {
'fade_for_StretchToFit': fade_for_StretchToFit,
'max_stretch_ratio': max_stretch_ratio,
'min_stretch_ratio': min_stretch_ratio,
'timing_tolerance': timing_tolerance,
}
result = engine_instance.processor.process_srt_content(
srt_content=srt_content,
voice_mapping=voice_mapping,
seed=seed,
timing_mode=timing_mode,
timing_params=timing_params,
enable_audio_cache=enable_audio_cache,
)
else:
raise ValueError(f"Unknown engine type: {engine_type}")
@@ -1702,6 +1772,8 @@ Hello! This is unified SRT TTS with character switching.
or "MOSS-TTSD Native Multi-Speaker Dialogue does not support this SRT input" in msg
):
raise
if engine_type == "audio_cpp":
raise
if isinstance(e, InterruptedError):
raise
error_msg = f"❌ TTS SRT generation failed: {e}"
+107 -1
View File
@@ -250,6 +250,19 @@ Back to the main narrator voice for the conclusion.""",
stable_params['dtype'] = config.get('dtype', 'auto')
stable_params['attention'] = config.get('attention', 'auto')
if engine_type == "audio_cpp":
# audio.cpp owns a persistent native server. Everything that changes
# that server/model session belongs in the instance cache identity;
# request-time sampling controls deliberately do not.
for key in (
'connection_mode', 'server_url', 'server_model_id', 'model_id',
'binary_path', 'model_path', 'model_roots', 'family',
'package_id', 'task', 'backend', 'device', 'device_index',
'threads', 'model_spec_override', 'load_options',
'session_options', 'show_server_console',
):
stable_params[key] = config.get(key)
# IndexTTS 2.0 and 2.5 are distinct checkpoints/backends. Every
# load-time option must participate in the processor cache key or
# changing the engine node can silently keep the old adapter alive.
@@ -746,6 +759,48 @@ Back to the main narrator voice for the conclusion.""",
return engine_instance
elif engine_type == "audio_cpp":
adapter_path = os.path.join(project_root, "engines", "adapters", "audio_cpp_adapter.py")
adapter_spec = importlib.util.spec_from_file_location(
"audio_cpp_adapter_module", adapter_path
)
if adapter_spec is None or adapter_spec.loader is None:
raise ImportError(f"Cannot load audio.cpp adapter from {adapter_path}")
adapter_module = importlib.util.module_from_spec(adapter_spec)
adapter_spec.loader.exec_module(adapter_module)
AudioCppEngineAdapter = adapter_module.AudioCppEngineAdapter
processor_path = os.path.join(nodes_dir, "audio_cpp", "audio_cpp_processor.py")
processor_spec = importlib.util.spec_from_file_location(
"audio_cpp_processor_module", processor_path
)
if processor_spec is None or processor_spec.loader is None:
raise ImportError(f"Cannot load audio.cpp processor from {processor_path}")
processor_module = importlib.util.module_from_spec(processor_spec)
processor_spec.loader.exec_module(processor_module)
AudioCppProcessor = processor_module.AudioCppProcessor
class AudioCppWrapper:
def __init__(self, cfg):
self.config = cfg.copy()
self.adapter = AudioCppEngineAdapter(self.config)
self.processor = AudioCppProcessor(self.adapter, self.config)
def update_config(self, new_config):
self.config = new_config.copy()
self.processor.update_config(self.config)
def check_interrupt(self):
if model_management.interrupt_processing:
raise InterruptedError("audio.cpp processing interrupted by user")
engine_instance = AudioCppWrapper(config)
import time
self._cached_engine_instances[cache_key] = {
'instance': engine_instance,
'timestamp': time.time(),
}
return engine_instance
elif engine_type == "step_audio_editx":
# Create Step Audio EditX wrapper instance
class StepAudioEditXWrapper:
@@ -1010,6 +1065,11 @@ Back to the main narrator voice for the conclusion.""",
if not engine_type:
raise ValueError("TTS engine missing engine_type")
capabilities = TTS_engine.get("capabilities", [])
if capabilities and "tts" not in capabilities:
raise ValueError(
f"Engine '{engine_type}' does not support TTS. Connect it to its compatible unified node."
)
if config.get("model_role") == "voice_design":
selected_model = config.get("model_variant") or config.get("model_name") or "selected model"
@@ -2055,6 +2115,52 @@ Back to the main narrator voice for the conclusion.""",
seed=seed
)
elif engine_type == "audio_cpp":
import re
voice_mapping = {}
if audio_tensor is not None or audio_path:
voice_mapping['narrator'] = {
'audio': audio_tensor,
'audio_path': audio_path,
'reference_text': reference_text or '',
}
audio_segments = engine_instance.processor.process_text(
text=text,
voice_mapping=voice_mapping,
seed=seed,
enable_chunking=enable_chunking,
max_chars_per_chunk=max_chars_per_chunk,
enable_audio_cache=enable_audio_cache,
)
audio_result, chunk_info = engine_instance.processor.combine_audio_segments(
segments=audio_segments,
method=chunk_combination_method,
silence_ms=silence_between_chunks_ms,
original_text=text,
return_info=True,
)
sample_rate = engine_instance.processor.sample_rate
if not sample_rate:
raise RuntimeError("audio.cpp returned no sample rate")
clean_text = re.sub(r'\[.*?\]', '', text)
duration = audio_result.shape[-1] / sample_rate if audio_result.numel() else 0.0
family = config.get('family') or config.get('server_model_id') or 'external model'
base_info = (
f"Generated {duration:.1f}s audio from {len(clean_text)} characters "
f"(audio.cpp {family}, {sample_rate} Hz, narrator: {char_display})"
)
base_info += "\n🎭 Character switching, pause tags, and per-segment parameters supported"
from utils.audio.chunk_timing import ChunkTimingHelper
generation_info = ChunkTimingHelper.enhance_generation_info(
f"✅ {base_info}", chunk_info
)
result = (
AudioProcessingUtils.format_for_comfyui(audio_result, sample_rate),
generation_info,
)
else:
raise ValueError(f"Unknown engine type: {engine_type}")
@@ -2090,7 +2196,7 @@ Back to the main narrator voice for the conclusion.""",
raise
if "MOSS LoRA/base model mismatch" in str(e):
raise
if engine_type == "index_tts":
if engine_type in {"index_tts", "audio_cpp"}:
raise
if isinstance(e, InterruptedError):
raise
+75 -18
View File
@@ -51,7 +51,7 @@ GLOBAL_RVC_ITERATION_CACHE = {}
class UnifiedVoiceChangerNode(BaseVCNode):
"""
Unified Voice Changer Node - Engine-agnostic voice conversion.
Currently supports ChatterBox, prepared for future RVC and other voice conversion engines.
Routes ChatterBox, CosyVoice, RVC, and compatible audio.cpp families.
Replaces ChatterBox VC node with engine-agnostic architecture.
"""
@@ -64,7 +64,7 @@ class UnifiedVoiceChangerNode(BaseVCNode):
return {
"required": {
"TTS_engine": ("TTS_ENGINE", {
"tooltip": "TTS/VC engine configuration. Supports ChatterBox TTS Engine, CosyVoice Engine, and RVC Engine for voice conversion."
"tooltip": "Engine configuration for source-to-target voice conversion. Supports ChatterBox, CosyVoice, RVC, and audio.cpp families whose panel shows Voice conversion (Chatterbox, VeVo2, or Seed-VC)."
}),
"source_audio": (any_typ, {
"tooltip": "The original voice audio you want to convert to sound like the target voice. Accepts AUDIO input or Character Voices node output."
@@ -574,7 +574,7 @@ class UnifiedVoiceChangerNode(BaseVCNode):
}
return engine_instance
elif engine_type == "cosyvoice":
elif engine_type == "cosyvoice":
# Import and create the CosyVoice VC processor
cosyvoice_vc_path = os.path.join(nodes_dir, "cosyvoice", "cosyvoice_vc_processor.py")
cosyvoice_vc_spec = importlib.util.spec_from_file_location("cosyvoice_vc_module", cosyvoice_vc_path)
@@ -589,10 +589,21 @@ class UnifiedVoiceChangerNode(BaseVCNode):
self._cached_engine_instances[cache_key] = {
'instance': engine_instance,
'timestamp': time.time()
}
return engine_instance
elif engine_type == "f5tts":
}
return engine_instance
elif engine_type == "audio_cpp":
from engines.adapters.audio_cpp_vc_adapter import AudioCppVoiceConversionAdapter
engine_instance = AudioCppVoiceConversionAdapter(config)
import time
self._cached_engine_instances[cache_key] = {
'instance': engine_instance,
'timestamp': time.time()
}
return engine_instance
elif engine_type == "f5tts":
# F5-TTS doesn't have voice conversion capability
raise ValueError("F5-TTS engine does not support voice conversion. Use ChatterBox or CosyVoice engine for voice conversion.")
@@ -831,14 +842,22 @@ class UnifiedVoiceChangerNode(BaseVCNode):
)
converted_chunk_audio = result[0]
elif engine_type == "cosyvoice":
elif engine_type == "cosyvoice":
# CosyVoice VC processor
result = engine_instance.convert_voice(
source_audio=chunk_audio_dict,
target_audio=target_audio,
refinement_passes=refinement_passes
)
converted_chunk_audio = result[0]
converted_chunk_audio = result[0]
elif engine_type == "audio_cpp":
result = engine_instance.convert_voice(
source_audio=chunk_audio_dict,
target_audio=target_audio,
refinement_passes=refinement_passes,
)
converted_chunk_audio = result[0]
else:
raise ValueError(f"Unsupported engine type for chunking: {engine_type}")
@@ -917,8 +936,13 @@ class UnifiedVoiceChangerNode(BaseVCNode):
print(f"🔄 Voice Changer: Starting {engine_type} voice conversion")
# Validate engine supports voice conversion
if engine_type not in ["chatterbox", "chatterbox_official_23lang", "rvc", "cosyvoice"]:
raise ValueError(f"Engine '{engine_type}' does not support voice conversion. Currently supported engines: ChatterBox, ChatterBox Official 23-Lang, RVC, CosyVoice")
if engine_type not in ["chatterbox", "chatterbox_official_23lang", "rvc", "cosyvoice", "audio_cpp"]:
raise ValueError(f"Engine '{engine_type}' does not support voice conversion. Currently supported engines: ChatterBox, ChatterBox Official 23-Lang, RVC, CosyVoice, audio.cpp")
if engine_type == "audio_cpp" and "voice_conversion" not in TTS_engine.get("capabilities", []):
family = config.get("family", "selected family")
raise ValueError(
f"audio.cpp family '{family}' does not map to the Suite's source/target Voice Changer contract"
)
# Extract audio data from flexible inputs (support both AUDIO and NARRATOR_VOICE types)
processed_source_audio = self._extract_audio_from_input(source_audio, "source_audio")
@@ -1079,7 +1103,7 @@ class UnifiedVoiceChangerNode(BaseVCNode):
f"Conversion completed successfully"
)
elif engine_type == "cosyvoice":
elif engine_type == "cosyvoice":
# CosyVoice voice conversion
print(f"🔄 Voice Changer: Using CosyVoice3 for voice conversion")
@@ -1120,12 +1144,45 @@ class UnifiedVoiceChangerNode(BaseVCNode):
)
# Add unified wrapper info
conversion_info = (
f"🔄 Voice Changer (Unified) - COSYVOICE3 Engine:\n"
f"{conversion_info}"
)
else:
conversion_info = (
f"🔄 Voice Changer (Unified) - COSYVOICE3 Engine:\n"
f"{conversion_info}"
)
elif engine_type == "audio_cpp":
if len(source_chunks) > 1:
converted_waveform, output_sample_rate = self._process_chunks_with_conversion(
source_chunks,
processed_narrator_target,
engine_instance,
engine_type,
refinement_passes,
config,
source_sample_rate,
)
converted_audio = {
"waveform": converted_waveform,
"sample_rate": output_sample_rate,
}
conversion_info = (
f"Model family: {config.get('family', 'external')}\n"
f"Chunks: {len(source_chunks)} ({chunk_method}, {max_chunk_duration}s max)\n"
f"Refinement passes: {refinement_passes}\n"
f"Output sample rate: {output_sample_rate} Hz\n"
"Conversion completed successfully"
)
else:
converted_audio, conversion_info = engine_instance.convert_voice(
source_audio=processed_source_audio,
target_audio=processed_narrator_target,
refinement_passes=refinement_passes,
)
conversion_info = (
"🔄 Voice Changer (Unified) - AUDIO.CPP Engine:\n"
f"{conversion_info}"
)
else:
# Future engines will be handled here
raise ValueError(f"Engine type '{engine_type}' voice conversion not yet implemented")
+66
View File
@@ -0,0 +1,66 @@
import sys
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp.capabilities import (
get_package_dependencies,
get_capability,
load_capabilities,
public_capabilities,
validate_voice_reference,
)
from utils.audio_cpp.catalog import load_catalog
@pytest.mark.unit
def test_capability_overlay_covers_the_pinned_catalog():
capabilities = load_capabilities()
assert set(capabilities) == set(load_catalog().families)
assert capabilities["vibevoice"]["native_multi_speaker"] == {
"supported": True,
"max_speakers": 4,
"suite_status": "partial",
}
assert capabilities["vibevoice_asr"]["asr_features"] == {
"diarization": "native",
"timing": "native_segment",
}
assert capabilities["nemotron_asr"]["asr_features"] == {
"diarization": "none",
"timing": "native_word",
}
assert capabilities["qwen3_asr"]["asr_features"]["timing"] == "optional_forced_aligner"
assert capabilities["voxtral_realtime"]["asr_features"] == {
"diarization": "none",
"timing": "none",
}
public = public_capabilities()
assert set(public["packages"]) == set(load_catalog().packages)
assert public["packages"]["qwen3_tts_1_7b_base_q8_0"]["estimated_download_bytes"] == 2695175104
mio = public["packages"]["miotts_1_7b_q8_0"]
assert mio["dependencies"] == ["miocodec_q8_0"]
assert mio["estimated_download_bytes"] == 2496393216
assert get_package_dependencies("miotts_1_7b_q8_0")[0]["session_option"] == "miotts.codec_model_path"
@pytest.mark.unit
def test_glm_requires_audio_and_matching_transcript():
with pytest.raises(ValueError, match="requires reference audio"):
validate_voice_reference("glm_tts", {}, "Alice")
with pytest.raises(ValueError, match="requires the transcript"):
validate_voice_reference(
"glm_tts",
{"audio": {"waveform": object(), "sample_rate": 24000}},
"Alice",
)
@pytest.mark.unit
def test_optional_reference_family_accepts_default_voice():
validate_voice_reference("pocket_tts", {}, "narrator")
assert get_capability("supertonic")["built_in_voices"] is True
+76
View File
@@ -0,0 +1,76 @@
from pathlib import Path
import sys
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp.catalog import (
AUDIO_CPP_RELEASE_VERSION,
CatalogError,
family_choices,
get_model_specs_dir,
load_catalog,
package_choices,
recommended_package,
resolve_task,
)
@pytest.mark.unit
def test_pinned_release_catalog_has_exact_suite_compatible_surface():
catalog = load_catalog()
assert AUDIO_CPP_RELEASE_VERSION == "0.5.1"
assert len(catalog.families) == 32
assert len(catalog.packages) == 96
assert set(family_choices()) == set(catalog.families)
assert len(package_choices()) == 96
assert set(path.name for path in get_model_specs_dir().glob("*.json")) == {
family.spec_filename for family in catalog.families.values()
}
assert "vevo2" in catalog.families
assert catalog.family("vevo2").runtime_tasks == ("tts", "vc", "s2s", "svc")
assert catalog.family("qwen3_asr").runtime_tasks == ("asr",)
assert catalog.family("seed_vc").runtime_tasks == ("vc", "svc")
@pytest.mark.unit
def test_catalog_merges_package_download_defaults_and_maps_local_paths():
catalog = load_catalog()
package = catalog.package("chatterbox_q8_0")
assert package.repo == "audio-cpp/audio.cpp-gguf"
assert package.revision == "main"
assert package.local_files == (Path("chatterbox-q8_0.gguf"),)
assert recommended_package("chatterbox") == "chatterbox_q8_0"
# Upstream release-0.5.1 uses strip_prefix="." here. It means no strip,
# not a literal directory named dot.
assert catalog.package("vietneu_tts_v3_turbo_q8_0").local_files == (Path("model.gguf"),)
@pytest.mark.unit
def test_resolve_task_uses_compiled_ids_and_specialized_package_semantics():
assert resolve_task("chatterbox", "chatterbox_q8_0", "clone") == "clon"
assert (
resolve_task("qwen3_tts", "qwen3_tts_1_7b_voicedesign_q8_0", "auto") == "vdes"
)
assert (
resolve_task("irodori_tts", "irodori_tts_600m_v3_voicedesign_f16", "auto") == "vdes"
)
assert resolve_task("qwen3_tts", "qwen3_tts_1_7b_base_q8_0", "auto") == "tts"
assert resolve_task("pocket_tts", "pocket_tts_english_q8_0", "clone") == "tts"
with pytest.raises(CatalogError, match="does not belong"):
resolve_task("chatterbox", "vevo2_q8_0", "auto")
@pytest.mark.unit
def test_every_package_maps_to_safe_relative_files():
for package in load_catalog().packages.values():
assert package.local_files
for path in package.local_files:
assert not path.is_absolute()
assert ".." not in path.parts
+103
View File
@@ -0,0 +1,103 @@
import io
from pathlib import Path
import sys
import urllib.error
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp.catalog import load_catalog
from utils.audio_cpp.downloader import AudioCppDownloadError, install_package, package_download_size
from utils.audio_cpp.discovery import package_install_path
class FakeResponse(io.BytesIO):
def __init__(self, payload, content_length=None):
super().__init__(payload)
self.status = 200
self.headers = {
"Content-Length": str(len(payload) if content_length is None else content_length)
}
@pytest.mark.unit
def test_package_download_size_uses_hf_metadata_without_downloading():
package = load_catalog().package("pocket_tts_english_q8_0")
requests = []
def opener(request, timeout):
requests.append(request)
return FakeResponse(b"", content_length=123_456)
assert package_download_size(package, token="secret-token", opener=opener) == 123_456
assert requests[0].method == "HEAD"
assert requests[0].get_header("Authorization") == "Bearer secret-token"
@pytest.mark.unit
def test_direct_hf_download_uses_auth_staging_and_nested_atomic_publish(tmp_path):
package = load_catalog().package("pocket_tts_english_q8_0")
payload = b"complete-gguf"
requests = []
def opener(request, timeout):
requests.append((request, timeout))
return FakeResponse(payload)
result = install_package(package, tmp_path, token="secret-token", opener=opener)
target = package_install_path(package, tmp_path)
assert result.path == target
assert (target / package.local_files[0]).read_bytes() == payload
assert requests[0][0].get_header("Authorization") == "Bearer secret-token"
assert "huggingface.co/audio-cpp/audio.cpp-gguf/resolve/main/" in requests[0][0].full_url
assert not list(target.parent.glob("*.staging"))
@pytest.mark.unit
def test_incomplete_http_response_never_publishes_package(tmp_path):
package = load_catalog().package("chatterbox_q8_0")
def opener(request, timeout):
return FakeResponse(b"short", content_length=100)
with pytest.raises(AudioCppDownloadError, match="Incomplete download"):
install_package(package, tmp_path, opener=opener)
assert not package_install_path(package, tmp_path).exists()
@pytest.mark.unit
def test_install_preserves_sibling_precision_in_shared_target(tmp_path):
package = load_catalog().package("chatterbox_q8_0")
sibling_package = load_catalog().package("chatterbox_f16")
target = package_install_path(package, tmp_path)
sibling = package_install_path(sibling_package, tmp_path)
target.mkdir(parents=True)
sibling_file = sibling / sibling_package.local_files[0]
sibling_file.write_bytes(b"keep")
result = install_package(
package,
tmp_path,
opener=lambda request, timeout: FakeResponse(b"new"),
)
assert (result.path / package.local_files[0]).read_bytes() == b"new"
assert sibling_file.read_bytes() == b"keep"
assert not list(target.parent.glob("*.backup"))
@pytest.mark.unit
def test_hf_auth_failure_is_actionable_and_leaves_no_target(tmp_path):
package = load_catalog().package("chatterbox_q8_0")
def opener(request, timeout):
raise urllib.error.HTTPError(request.full_url, 401, "Unauthorized", {}, None)
with pytest.raises(AudioCppDownloadError, match="HF_TOKEN"):
install_package(package, tmp_path, opener=opener)
assert not package_install_path(package, tmp_path).exists()
+346
View File
@@ -0,0 +1,346 @@
"""Focused tests for the audio.cpp adapter, processors, and engine node."""
from __future__ import annotations
import importlib.util
from pathlib import Path
from types import SimpleNamespace
import pytest
import torch
PROJECT_ROOT = Path(__file__).resolve().parents[2]
def _load_module(name, relative_path):
spec = importlib.util.spec_from_file_location(name, PROJECT_ROOT / relative_path)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
adapter_module = _load_module("audio_cpp_adapter_test_module", "engines/adapters/audio_cpp_adapter.py")
processor_module = _load_module("audio_cpp_processor_test_module", "nodes/audio_cpp/audio_cpp_processor.py")
srt_module = _load_module("audio_cpp_srt_test_module", "nodes/audio_cpp/audio_cpp_srt_processor.py")
node_module = _load_module("audio_cpp_engine_node_test_module", "nodes/engines/audio_cpp_engine_node.py")
class _FakeSession:
def __init__(self, sample_rate=32000):
self.sample_rate = sample_rate
self.requests = []
self.owned = True
self.endpoint = ""
self.model_id = "owned-test-model"
self.family = "qwen3_tts"
self.config = {"task": "tts"}
def run(self, request):
self.requests.append(dict(request))
# Owned sessions receive a random HTTP endpoint only after the server
# starts. That transient port must not change the audio cache identity.
self.endpoint = "http://127.0.0.1:54321"
if request.get("voice_ref"):
from pathlib import Path
assert Path(request["voice_ref"]).is_absolute()
assert Path(request["voice_ref"]).is_file()
return SimpleNamespace(
waveform=torch.ones(1, self.sample_rate // 10),
sample_rate=self.sample_rate,
named_audio={},
)
def test_adapter_caches_real_sample_rate_and_cleans_reference(monkeypatch, tmp_path):
adapter_module.get_audio_cache().clear_cache()
adapter_module._CACHE_SAMPLE_RATES.clear()
session = _FakeSession(sample_rate=32000)
monkeypatch.setattr(adapter_module, "_get_session", lambda config: session)
created = []
def fake_save(waveform, sample_rate):
path = tmp_path / f"reference-{len(created)}.wav"
path.write_bytes(b"temporary")
created.append(path)
return str(path)
monkeypatch.setattr(
adapter_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
)
adapter = adapter_module.AudioCppEngineAdapter(
{"family": "qwen3_tts", "package_id": "qwen3_tts_1_7b_base_q8_0", "task": "tts"}
)
voice = {"audio": {"waveform": torch.zeros(1, 80), "sample_rate": 16000}}
first, first_rate = adapter.generate_single("hello", voice, seed=7)
second, second_rate = adapter.generate_single("hello", voice, seed=7)
assert first_rate == second_rate == 32000
assert torch.equal(first, second)
assert len(session.requests) == 1
assert session.requests[0]["seed"] == "7"
assert len(created) == 1
assert created[0].exists()
adapter.close()
assert all(not path.exists() for path in created)
class _ProcessorAdapter:
def __init__(self, sample_rate=24000):
self.sample_rate = sample_rate
self.config = {}
def update_config(self, config):
self.config = dict(config)
def generate_single(self, **kwargs):
return torch.ones(1, self.sample_rate // 10), self.sample_rate
def test_processor_materializes_leading_pause_at_response_rate(monkeypatch):
segment = SimpleNamespace(
text="[pause:0.01] hello",
character="narrator",
parameters={},
language=None,
explicit_language=False,
)
monkeypatch.setattr(processor_module.AudioCppProcessor, "_setup_character_parser", lambda self, text: None)
monkeypatch.setattr(
processor_module.character_parser,
"parse_text_segments",
lambda text, engine_type=None: [segment],
)
monkeypatch.setattr(processor_module, "get_character_mapping", lambda *args, **kwargs: {})
processor = processor_module.AudioCppProcessor(_ProcessorAdapter(32000), {"language": "auto"})
records = processor.process_text(
"ignored", {}, seed=1, enable_chunking=False, show_text_logging=False
)
assert processor.sample_rate == 32000
assert records[0]["sample_rate"] == 32000
assert records[0]["waveform"].shape == (1, 320)
assert records[1]["sample_rate"] == 32000
def test_processor_uses_glm_transcripts_and_resets_rate_between_generations(monkeypatch):
segment = SimpleNamespace(
text="hello",
character="Alice",
parameters={},
language=None,
explicit_language=False,
)
monkeypatch.setattr(
processor_module.AudioCppProcessor, "_setup_character_parser", lambda self, text: None
)
monkeypatch.setattr(
processor_module.character_parser,
"parse_text_segments",
lambda text, engine_type=None: [segment],
)
discovery_modes = []
def mapping(characters, engine_type):
discovery_modes.append(engine_type)
return {"Alice": ("alice.wav", "matching transcript")}
monkeypatch.setattr(processor_module, "get_character_mapping", mapping)
adapter = _ProcessorAdapter(24000)
processor = processor_module.AudioCppProcessor(adapter, {"family": "glm_tts"})
processor.process_text("first", {}, seed=1, enable_chunking=False, show_text_logging=False)
adapter.sample_rate = 32000
processor.process_text("second", {}, seed=1, enable_chunking=False, show_text_logging=False)
assert discovery_modes == ["audio_and_text", "audio_and_text"]
assert processor.sample_rate == 32000
def test_processor_rejects_mixed_response_rates():
processor = processor_module.AudioCppProcessor(_ProcessorAdapter(), {})
segments = [
{"waveform": torch.zeros(1, 8), "sample_rate": 24000, "text": "a"},
{"waveform": torch.zeros(1, 8), "sample_rate": 32000, "text": "b"},
]
with pytest.raises(RuntimeError, match="inconsistent sample rates"):
processor.combine_audio_segments(segments)
def test_engine_node_resolves_owned_package_task(monkeypatch):
monkeypatch.setattr(node_module, "_recommended_package", lambda family: "design-package")
monkeypatch.setattr(node_module, "_validate_package", lambda family, package: None)
monkeypatch.setattr(node_module, "_resolve_task", lambda family, package, task: "vdes")
engine = node_module.AudioCppEngineNode().create_engine_config(
"managed", "qwen3_tts", "auto", "auto", "cuda", 0, 4, "auto"
)[0]
assert engine["config"]["package_id"] == "design-package"
assert engine["config"]["task"] == "vdes"
assert engine["config"]["threads"] == 4
assert engine["capabilities"] == ["tts", "voice_design"]
def test_engine_node_keeps_external_server_task_authoritative():
engine = node_module.AudioCppEngineNode().create_engine_config(
"external_server",
"qwen3_tts",
"auto",
"auto",
"cpu",
0,
4,
"auto",
server_url="http://127.0.0.1:8080",
)[0]
assert engine["config"]["task"] == "auto"
assert engine["config"]["package_id"] == "auto"
def test_engine_node_uses_pinned_catalog_contract():
inputs = node_module.AudioCppEngineNode.INPUT_TYPES()
assert "qwen3_tts" in inputs["required"]["family"][0]
assert "qwen3_tts_1_7b_base_q8_0" in inputs["required"]["package_id"][0]
engine = node_module.AudioCppEngineNode().create_engine_config(
"managed", "qwen3_tts", "auto", "auto", "cpu", 0, 4, "auto"
)[0]
assert engine["config"]["package_id"] == "qwen3_tts_1_7b_base_q8_0"
assert engine["config"]["task"] == "tts"
def test_unified_nodes_construct_audio_cpp_processors_without_nodes_package_collision():
text_module = _load_module(
"audio_cpp_unified_text_test_module", "nodes/unified/tts_text_node.py"
)
srt_unified_module = _load_module(
"audio_cpp_unified_srt_test_module", "nodes/unified/tts_srt_node.py"
)
engine = node_module.AudioCppEngineNode().create_engine_config(
"external_server",
"pocket_tts",
"auto",
"auto",
"cpu",
0,
4,
"auto",
server_url="http://127.0.0.1:9999",
model_id="wiring-only",
)[0]
text_wrapper = text_module.UnifiedTTSTextNode()._create_proper_engine_node_instance(engine)
srt_wrapper = srt_unified_module.UnifiedTTSSRTNode()._create_proper_engine_node_instance(engine)
assert type(text_wrapper.adapter).__name__ == "AudioCppEngineAdapter"
assert type(text_wrapper.processor).__name__ == "AudioCppProcessor"
assert type(srt_wrapper.processor).__name__ == "AudioCppSRTProcessor"
def test_unified_nodes_surface_audio_cpp_runtime_errors(monkeypatch):
from utils.audio_cpp import session as session_module
text_module = _load_module(
"audio_cpp_unified_text_error_test_module", "nodes/unified/tts_text_node.py"
)
srt_unified_module = _load_module(
"audio_cpp_unified_srt_error_test_module", "nodes/unified/tts_srt_node.py"
)
engine = node_module.AudioCppEngineNode().create_engine_config(
"external_server",
"pocket_tts",
"auto",
"auto",
"cpu",
0,
4,
"auto",
server_url="http://127.0.0.1:9999",
model_id="error-only",
)[0]
def fail_session(config):
raise RuntimeError("visible audio.cpp failure")
monkeypatch.setattr(session_module, "get_audio_cpp_session", fail_session)
with pytest.raises(RuntimeError, match="visible audio.cpp failure"):
text_module.UnifiedTTSTextNode().generate_speech(
engine, "hello", "none", 1, enable_chunking=False, enable_audio_cache=False
)
with pytest.raises(RuntimeError, match="visible audio.cpp failure"):
srt_unified_module.UnifiedTTSSRTNode().generate_srt_speech(
engine,
"1\n00:00:00,000 --> 00:00:01,000\nhello",
"none",
1,
"concatenate",
enable_audio_cache=False,
)
class _Subtitle:
def __init__(self, sequence, text, start, end):
self.sequence = sequence
self.text = text
self.start_time = start
self.end_time = end
self.duration = end - start
class _SRTTextProcessor:
sample_rate = 16000
def reset_sample_rate(self):
return None
def process_text(self, **kwargs):
return [{"waveform": torch.ones(1, 8000), "sample_rate": 16000, "text": kwargs["text"]}]
def combine_audio_segments(self, records, **kwargs):
return records[0]["waveform"]
def test_srt_delays_blank_cue_until_dynamic_rate_is_known(monkeypatch):
subtitles = [_Subtitle(1, "", 0.0, 0.25), _Subtitle(2, "hello", 0.25, 0.75)]
instance = srt_module.AudioCppSRTProcessor.__new__(srt_module.AudioCppSRTProcessor)
instance.config = {}
instance._processor = _SRTTextProcessor()
instance.SRTParser = lambda: SimpleNamespace(
parse_srt_content=lambda content, allow_overlaps: subtitles
)
monkeypatch.setattr(instance, "_check_interrupt", lambda *args: None)
monkeypatch.setattr(srt_module.SRTOverlapHandler, "detect_overlaps", lambda items: False)
monkeypatch.setattr(
srt_module.SRTOverlapHandler,
"handle_smart_natural_fallback",
lambda mode, overlaps, label: (mode, False),
)
captured = {}
def fake_assemble(audio, subs, mode, params, rate):
captured["segments"] = audio
return torch.cat(audio, dim=-1), None, None
monkeypatch.setattr(instance, "_assemble", fake_assemble)
monkeypatch.setattr(
srt_module,
"SRTReportGenerator",
lambda: SimpleNamespace(
generate_timing_report=lambda *args: "report",
generate_adjusted_srt_string=lambda *args: "adjusted",
),
)
audio, _, report, adjusted = instance.process_srt_content(
"unused", {}, 0, "concatenate", {}, enable_audio_cache=False
)
assert captured["segments"][0].shape[-1] == 4000
assert audio["sample_rate"] == 16000
assert audio["waveform"].shape == (1, 1, 12000)
assert (report, adjusted) == ("report", "adjusted")
+260
View File
@@ -0,0 +1,260 @@
"""No-model tests for audio.cpp ASR and unified voice-conversion contracts."""
from __future__ import annotations
import importlib.util
from pathlib import Path
from types import SimpleNamespace
import pytest
import torch
from utils.asr.types import ASRRequest
from utils.audio_cpp import session as session_module
PROJECT_ROOT = Path(__file__).resolve().parents[2]
def _load_module(name: str, relative_path: str):
spec = importlib.util.spec_from_file_location(name, PROJECT_ROOT / relative_path)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
asr_module = _load_module(
"audio_cpp_asr_adapter_test_module", "engines/adapters/asr_audio_cpp_adapter.py"
)
vc_module = _load_module(
"audio_cpp_vc_adapter_test_module", "engines/adapters/audio_cpp_vc_adapter.py"
)
node_module = _load_module(
"audio_cpp_multitask_node_test_module", "nodes/engines/audio_cpp_engine_node.py"
)
def _fake_save_factory(tmp_path):
paths = []
def save(_waveform, _sample_rate):
path = tmp_path / f"audio-{len(paths)}.wav"
path.write_bytes(b"wav")
paths.append(path)
return str(path)
return paths, save
@pytest.mark.unit
def test_engine_node_advertises_asr_and_vc_consumers():
asr_engine = node_module.AudioCppEngineNode().create_engine_config(
"managed", "qwen3_asr", "auto", "auto", "cpu", 0, 4, "auto"
)[0]
vc_engine = node_module.AudioCppEngineNode().create_engine_config(
"managed", "seed_vc", "auto", "auto", "cpu", 0, 4, "auto"
)[0]
assert asr_engine["config"]["task"] == "asr"
assert asr_engine["capabilities"] == ["asr"]
assert vc_engine["config"]["task"] == "vc"
assert vc_engine["capabilities"] == ["voice_conversion"]
@pytest.mark.unit
def test_asr_adapter_normalizes_words_and_speaker_turns(monkeypatch, tmp_path):
paths, fake_save = _fake_save_factory(tmp_path)
monkeypatch.setattr(
asr_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
)
class FakeSession:
task = "asr"
model_id = "vibe-asr"
def run(self, request):
assert Path(request["audio"]).is_absolute()
assert Path(request["audio"]).is_file()
return SimpleNamespace(raw={
"text": "hello world",
"language": "en",
"words": [
{"word": "hello", "start_sample": 0, "end_sample": 8000},
{"word": "world", "start_sample": 8000, "end_sample": 16000},
],
"speaker_turns": [
{
"start_sample": 0,
"end_sample": 16000,
"speaker_id": "Speaker 1",
"text": "hello world",
}
],
})
monkeypatch.setattr(session_module, "get_audio_cpp_session", lambda _config: FakeSession())
adapter = asr_module.AudioCppASREngineAdapter({
"engine_type": "audio_cpp",
"config": {"family": "vibevoice_asr", "connection_mode": "external_server"},
})
result = adapter.transcribe(ASRRequest(
audio={"waveform": torch.zeros(1, 1, 16000), "sample_rate": 16000},
timestamps="word",
diarization=True,
chunk_size=0,
))
assert result.text == "[Speaker 1] hello world"
assert result.language == "en"
assert result.segments[0].speaker == "Speaker 1"
assert [word.text for word in result.segments[0].words] == ["hello", "world"]
assert paths and not paths[0].exists()
@pytest.mark.unit
def test_asr_adapter_uses_suite_chunking_and_deduplicates_overlap(monkeypatch, tmp_path):
paths, fake_save = _fake_save_factory(tmp_path)
monkeypatch.setattr(
asr_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
)
class FakeSession:
task = "asr"
model_id = "nemotron-asr"
owned = True
def __init__(self):
self.requests = []
self.restarts = 0
self.texts = iter((
"one two three",
"three four five",
"five six seven",
))
def restart_owned_runtime(self):
self.restarts += 1
def run(self, request):
self.requests.append(request)
assert Path(request["audio"]).is_file()
return SimpleNamespace(raw={"text": next(self.texts), "language": "en"})
fake_session = FakeSession()
monkeypatch.setattr(
session_module, "get_audio_cpp_session", lambda _config: fake_session
)
adapter = asr_module.AudioCppASREngineAdapter({
"engine_type": "audio_cpp",
"config": {"family": "nemotron_asr", "connection_mode": "external_server"},
})
result = adapter.transcribe(ASRRequest(
audio={"waveform": torch.zeros(1, 1, 80), "sample_rate": 10},
chunk_size=4,
overlap=2,
))
assert result.text == "one two three four five six seven"
assert len(fake_session.requests) == 3
assert fake_session.restarts == 2
assert result.raw["timing"]["suite_chunks"] == 3
assert any("Suite-side ASR chunking" in note for note in result.raw["notes"])
assert [chunk["text"] for chunk in result.raw["chunks"]] == [
"one two three",
"three four five",
"five six seven",
]
assert [(chunk["start"], chunk["end"]) for chunk in result.raw["chunks"]] == [
(0.0, 4.0),
(2.0, 6.0),
(4.0, 8.0),
]
assert paths and all(not path.exists() for path in paths)
@pytest.mark.unit
def test_vibevoice_diarization_keeps_native_chunking(monkeypatch, tmp_path):
paths, fake_save = _fake_save_factory(tmp_path)
monkeypatch.setattr(
asr_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
)
captured = {}
class FakeSession:
task = "asr"
model_id = "vibe-asr"
def run(self, request):
captured.update(request)
return SimpleNamespace(raw={
"text": "hello",
"speaker_turns": [{
"start_sample": 0,
"end_sample": 16000,
"speaker_id": "1",
"text": "hello",
}],
})
monkeypatch.setattr(
session_module, "get_audio_cpp_session", lambda _config: FakeSession()
)
adapter = asr_module.AudioCppASREngineAdapter({
"engine_type": "audio_cpp",
"config": {"family": "vibevoice_asr", "connection_mode": "external_server"},
})
result = adapter.transcribe(ASRRequest(
audio={"waveform": torch.zeros(1, 1, 16000), "sample_rate": 16000},
diarization=True,
chunk_size=30,
overlap=2,
))
assert captured["options"]["audio_chunk_mode"] == "fixed"
assert captured["options"]["audio_chunk_seconds"] == 30
assert result.text == "[Speaker 1] hello"
assert len(paths) == 1 and not paths[0].exists()
@pytest.mark.unit
def test_vc_adapter_forces_vc_task_and_uses_source_target_audio(monkeypatch, tmp_path):
paths, fake_save = _fake_save_factory(tmp_path)
monkeypatch.setattr(
vc_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
)
captured = {}
class FakeSession:
task = "vc"
model_id = "seed-vc"
family = "seed_vc"
def run(self, request):
captured.update(request)
assert Path(request["audio"]).is_file()
assert Path(request["voice_ref"]).is_file()
return SimpleNamespace(
waveform=torch.ones(1, 2400),
sample_rate=24000,
named_audio={},
)
def fake_session(config):
assert config["requested_task"] == "vc"
assert config["task"] == "vc"
return FakeSession()
monkeypatch.setattr(session_module, "get_audio_cpp_session", fake_session)
adapter = vc_module.AudioCppVoiceConversionAdapter({
"family": "seed_vc",
"connection_mode": "managed",
})
audio = {"waveform": torch.zeros(1, 1, 1600), "sample_rate": 16000}
converted, info = adapter.convert_voice(audio, audio)
assert converted["waveform"].shape == (1, 1, 2400)
assert converted["sample_rate"] == 24000
assert captured["source_audio"] == captured["audio"]
assert captured["target_voice"] == captured["voice_ref"]
assert "Seed-VC" not in info or "seed_vc" in info
assert paths and all(not path.exists() for path in paths)
+332
View File
@@ -0,0 +1,332 @@
from pathlib import Path
from types import SimpleNamespace
import sys
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp import resolver
from utils.audio_cpp.catalog import load_catalog
from utils.audio_cpp.discovery import package_install_path
from utils.audio_cpp.runtime_installer import runtime_install_path
from utils.audio_cpp.settings import AudioCppSettings
@pytest.mark.unit
def test_auto_reuses_machine_external_server_without_loading_owned_dependencies(monkeypatch):
monkeypatch.setattr(
resolver,
"load_settings",
lambda: AudioCppSettings(
connection_mode="external",
external_server_url="HTTP://127.0.0.1:18080/",
executable_path="C:/ignored/audiocpp_server.exe",
),
)
monkeypatch.setattr(
resolver,
"_catalog_module",
lambda: (_ for _ in ()).throw(AssertionError("external mode loaded the catalog")),
)
monkeypatch.setattr(
resolver,
"_downloader_module",
lambda: (_ for _ in ()).throw(AssertionError("external mode loaded a downloader")),
)
monkeypatch.setattr(
resolver,
"_runtime_installer_module",
lambda: (_ for _ in ()).throw(AssertionError("external mode loaded an installer")),
)
result = resolver.resolve_audio_cpp_config(
{
"config": {
"connection_mode": "auto",
"family": "qwen3_tts",
"package_id": "auto",
"task": "auto",
}
}
)
assert result["connection_mode"] == "external_server"
assert result["server_url"] == "http://127.0.0.1:18080"
assert result["binary_path"] == ""
assert result["model_path"] == ""
assert "model_id" not in result
assert result["task"] == "auto"
@pytest.mark.unit
def test_owned_explicit_paths_resolve_recommended_package_task_and_cuda(monkeypatch, tmp_path):
binary = tmp_path / "audiocpp_server.exe"
binary.write_bytes(b"exe")
model = tmp_path / "Qwen-VoiceDesign"
model.mkdir()
monkeypatch.setattr(resolver, "load_settings", lambda: AudioCppSettings())
monkeypatch.setattr(resolver, "_cuda_available", lambda: True)
result = resolver.resolve_audio_cpp_config(
{
"connection_mode": "managed",
"family": "qwen3_tts",
"package_id": "qwen3_tts_1_7b_voicedesign_q8_0",
"requested_task": "auto",
"backend": "auto",
"device": "cuda:2",
"binary_path": str(binary),
"model_path": str(model),
"model_id": "My Qwen model",
}
)
assert result["connection_mode"] == "owned_process"
assert result["package_id"] == "qwen3_tts_1_7b_voicedesign_q8_0"
assert result["task"] == "vdes"
assert result["backend"] == "cuda"
assert result["device_index"] == 2
assert result["binary_path"] == str(binary.resolve())
assert result["model_path"] == str(model.resolve())
assert result["model_id"] == "My-Qwen-model"
@pytest.mark.unit
def test_owned_reuses_external_model_root_and_configured_runtime_root(monkeypatch, tmp_path):
external_models = tmp_path / "existing-audio-cpp" / "models"
managed_models = tmp_path / "suite" / "audio.cpp" / "models"
configured_runtime = tmp_path / "existing-audio-cpp" / "runtime"
settings = AudioCppSettings(
connection_mode="managed",
model_roots=(str(external_models),),
managed_model_root=str(managed_models),
runtime_root=str(configured_runtime),
runtime_backend="cpu",
)
package = load_catalog().package("chatterbox_q8_0")
installed_model = package_install_path(package, external_models)
installed_model.mkdir(parents=True)
(installed_model / package.local_files[0]).write_bytes(b"gguf")
installed_binary = runtime_install_path(configured_runtime, "cpu") / "audiocpp_server.exe"
installed_binary.parent.mkdir(parents=True)
installed_binary.write_bytes(b"exe")
monkeypatch.setattr(resolver, "load_settings", lambda: settings)
result = resolver.resolve_audio_cpp_config(
{
"connection_mode": "auto",
"family": "chatterbox",
"package_id": "auto",
"task": "auto",
"backend": "auto",
"auto_download_model": False,
"auto_download_runtime": False,
}
)
assert result["package_id"] == "chatterbox_q8_0"
assert result["task"] == "clon"
assert result["backend"] == "cpu"
assert result["model_path"] == str(installed_model.resolve())
assert result["binary_path"] == str(installed_binary.resolve())
assert not managed_models.exists()
@pytest.mark.unit
def test_missing_assets_download_only_to_managed_roots(monkeypatch, tmp_path):
external_models = tmp_path / "external" / "models"
managed_models = tmp_path / "managed" / "audio.cpp" / "models"
settings = AudioCppSettings(
connection_mode="managed",
model_roots=(str(external_models),),
managed_model_root=str(managed_models),
runtime_backend="cpu",
)
calls = {}
real_downloader = resolver._downloader_module()
real_runtime = resolver._runtime_installer_module()
def install_package(package, root, catalog, progress=None):
calls["model_root"] = Path(root)
calls["model_progress"] = progress
target = package_install_path(package, root)
target.mkdir(parents=True)
(target / package.local_files[0]).write_bytes(b"gguf")
return SimpleNamespace(path=target, bytes_downloaded=4)
def install_runtime(root, backend, progress=None):
calls["runtime_root"] = Path(root)
calls["backend"] = backend
calls["runtime_progress"] = progress
executable = runtime_install_path(root, backend) / "audiocpp_server.exe"
executable.parent.mkdir(parents=True)
executable.write_bytes(b"exe")
return SimpleNamespace(executable=executable)
monkeypatch.setattr(resolver, "load_settings", lambda: settings)
monkeypatch.setattr(
resolver,
"_downloader_module",
lambda: SimpleNamespace(install_package=install_package),
)
monkeypatch.setattr(
resolver,
"_runtime_installer_module",
lambda: SimpleNamespace(
runtime_install_path=real_runtime.runtime_install_path,
get_runtime_manifest=real_runtime.get_runtime_manifest,
install_windows_runtime=install_runtime,
),
)
result = resolver.resolve_audio_cpp_config(
{
"connection_mode": "managed",
"family": "chatterbox",
"package_id": "chatterbox_q8_0",
"backend": "cpu",
"auto_download_model": True,
"auto_download_runtime": True,
}
)
assert calls["model_root"] == managed_models
assert calls["runtime_root"] == managed_models.parent / "runtime"
assert calls["backend"] == "cpu"
assert callable(calls["model_progress"])
assert callable(calls["runtime_progress"])
assert not external_models.exists()
assert result["model_path"].startswith(str(managed_models.resolve()))
assert result["binary_path"].startswith(str((managed_models.parent / "runtime").resolve()))
@pytest.mark.unit
def test_missing_model_without_permission_has_actionable_error(monkeypatch, tmp_path):
managed_models = tmp_path / "managed" / "models"
binary = tmp_path / "audiocpp_server.exe"
binary.write_bytes(b"exe")
monkeypatch.setattr(
resolver,
"load_settings",
lambda: AudioCppSettings(managed_model_root=str(managed_models)),
)
with pytest.raises(resolver.AudioCppResolutionError, match="enable auto_download_model"):
resolver.resolve_audio_cpp_config(
{
"connection_mode": "managed",
"family": "chatterbox",
"package_id": "chatterbox_q8_0",
"backend": "cpu",
"binary_path": str(binary),
"auto_download_model": False,
}
)
@pytest.mark.unit
def test_miotts_installs_codec_dependency_and_sets_absolute_session_path(monkeypatch, tmp_path):
managed_models = tmp_path / "managed" / "models"
catalog = resolver._catalog_module().load_catalog()
miotts = catalog.package("miotts_1_7b_q8_0")
miotts_dir = package_install_path(miotts, managed_models)
for relative in miotts.local_files:
target = miotts_dir / relative
target.parent.mkdir(parents=True, exist_ok=True)
target.write_bytes(b"miotts")
binary = tmp_path / "audiocpp_server.exe"
binary.write_bytes(b"exe")
installed = {}
def install_package(package, root, **_kwargs):
target = package_install_path(package, root)
for relative in package.local_files:
path = target / relative
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(b"codec")
installed[package.id] = target
return SimpleNamespace(path=target, bytes_downloaded=5)
monkeypatch.setattr(
resolver,
"load_settings",
lambda: AudioCppSettings(managed_model_root=str(managed_models)),
)
monkeypatch.setattr(
resolver,
"_downloader_module",
lambda: SimpleNamespace(install_package=install_package),
)
result = resolver.resolve_audio_cpp_config(
{
"connection_mode": "managed",
"family": "miotts",
"package_id": "miotts_1_7b_q8_0",
"backend": "cpu",
"binary_path": str(binary),
"auto_download_model": True,
}
)
assert "miocodec_q8_0" in installed
assert result["session_options"]["miotts.codec_model_path"] == str(
installed["miocodec_q8_0"].resolve()
)
@pytest.mark.unit
def test_owned_existing_binary_allows_explicit_hip_backend(monkeypatch, tmp_path):
binary = tmp_path / "audiocpp_server.exe"
binary.write_bytes(b"exe")
model = tmp_path / "model"
model.mkdir()
monkeypatch.setattr(resolver, "load_settings", lambda: AudioCppSettings())
result = resolver.resolve_audio_cpp_config(
{
"connection_mode": "existing_binary",
"family": "chatterbox",
"package_id": "chatterbox_q8_0",
"backend": "hip",
"binary_path": str(binary),
"model_path": str(model),
}
)
assert result["backend"] == "hip"
assert result["binary_path"] == str(binary.resolve())
@pytest.mark.unit
def test_cpu_backend_reuses_installed_cuda_profile_before_downloading_duplicate(
monkeypatch, tmp_path
):
model = tmp_path / "model"
model.mkdir()
runtime_root = tmp_path / "runtime"
cuda_binary = runtime_install_path(runtime_root, "cuda") / "audiocpp_server.exe"
cuda_binary.parent.mkdir(parents=True)
cuda_binary.write_bytes(b"cuda-exe")
monkeypatch.setattr(
resolver,
"load_settings",
lambda: AudioCppSettings(runtime_root=str(runtime_root), runtime_backend="cpu"),
)
result = resolver.resolve_audio_cpp_config(
{
"connection_mode": "managed",
"family": "chatterbox",
"package_id": "chatterbox_q8_0",
"model_path": str(model),
"backend": "auto",
"auto_download_runtime": False,
}
)
assert result["backend"] == "cpu"
assert result["binary_path"] == str(cuda_binary.resolve())
@@ -0,0 +1,145 @@
import hashlib
import io
from pathlib import Path
import sys
import zipfile
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp.runtime_installer import (
RuntimeAsset,
RuntimeInstallError,
RuntimeManifest,
get_runtime_manifest,
install_windows_runtime,
runtime_install_path,
)
class FakeResponse(io.BytesIO):
def __init__(self, payload):
super().__init__(payload)
self.status = 200
self.headers = {"Content-Length": str(len(payload))}
def make_zip(files):
stream = io.BytesIO()
with zipfile.ZipFile(stream, "w") as bundle:
for name, payload in files.items():
bundle.writestr(name, payload)
return stream.getvalue()
def fake_manifest(backend, payloads, required):
assets = tuple(
RuntimeAsset(
filename=name,
url=f"https://example.test/{name}",
size=len(payload),
sha256=hashlib.sha256(payload).hexdigest(),
)
for name, payload in payloads.items()
)
return RuntimeManifest(backend=backend, assets=assets, required_files=tuple(required))
@pytest.mark.unit
def test_pinned_windows_manifests_include_verified_cuda_runtime_asset():
cpu = get_runtime_manifest("cpu")
cuda = get_runtime_manifest("cuda")
assert len(cpu.assets) == 1
assert len(cuda.assets) == 2
assert cuda.assets[1].filename == "audiocpp-windows-cuda-runtime.zip"
assert cuda.assets[1].sha256 == (
"46016655aff8f050806d81efd0fe256c15b86527935bfb3896208d4cac6b5ff8"
)
assert "cublas64_13.dll" in cuda.required_files
@pytest.mark.unit
def test_cuda_runtime_installs_two_verified_archives_atomically(tmp_path):
executable_zip = make_zip(
{"audiocpp_server.exe": b"server", "audiocpp_cli.exe": b"cli"}
)
cuda_zip = make_zip({"cublas64_13.dll": b"cublas", "cufft64_12.dll": b"cufft"})
payloads = {"runtime.zip": executable_zip, "cuda.zip": cuda_zip}
manifest = fake_manifest(
"cuda",
payloads,
("audiocpp_server.exe", "audiocpp_cli.exe", "cublas64_13.dll", "cufft64_12.dll"),
)
def opener(request, timeout):
return FakeResponse(payloads[Path(request.full_url).name])
result = install_windows_runtime(
tmp_path,
"cuda",
manifest=manifest,
platform_name="win32",
opener=opener,
)
assert result.path == runtime_install_path(tmp_path, "cuda")
assert result.executable.read_bytes() == b"server"
assert (result.path / "cublas64_13.dll").read_bytes() == b"cublas"
assert not list(result.path.parent.glob("*.staging"))
@pytest.mark.unit
def test_runtime_hash_failure_does_not_publish_or_destroy_existing_target(tmp_path):
archive = make_zip({"audiocpp_server.exe": b"new", "audiocpp_cli.exe": b"cli"})
asset = RuntimeAsset(
filename="runtime.zip",
url="https://example.test/runtime.zip",
size=len(archive),
sha256="0" * 64,
)
manifest = RuntimeManifest(
backend="cpu",
assets=(asset,),
required_files=("audiocpp_server.exe", "audiocpp_cli.exe"),
)
target = runtime_install_path(tmp_path, "cpu")
target.mkdir(parents=True)
marker = target / "old.txt"
marker.write_bytes(b"old")
with pytest.raises(RuntimeInstallError, match="SHA256 mismatch"):
install_windows_runtime(
tmp_path,
"cpu",
overwrite=True,
manifest=manifest,
platform_name="win32",
opener=lambda request, timeout: FakeResponse(archive),
)
assert marker.read_bytes() == b"old"
assert not (target / "audiocpp_server.exe").exists()
@pytest.mark.unit
def test_runtime_rejects_archive_path_traversal(tmp_path):
archive = make_zip(
{"../escape.dll": b"bad", "audiocpp_server.exe": b"server", "audiocpp_cli.exe": b"cli"}
)
manifest = fake_manifest(
"cpu", {"runtime.zip": archive}, ("audiocpp_server.exe", "audiocpp_cli.exe")
)
with pytest.raises(RuntimeInstallError, match="Unsafe path"):
install_windows_runtime(
tmp_path,
"cpu",
manifest=manifest,
platform_name="win32",
opener=lambda request, timeout: FakeResponse(archive),
)
assert not (tmp_path / "escape.dll").exists()
@@ -0,0 +1,109 @@
import json
from pathlib import Path
import sys
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp.catalog import load_catalog
from utils.audio_cpp.discovery import (
find_installed_package,
package_install_path,
resolve_model,
resolve_model_roots,
)
from utils.audio_cpp.settings import AudioCppSettings, get_settings_path, load_settings, save_settings
class FakeFolderPaths:
def __init__(self, user_root, models_root, registry):
self.user_root = Path(user_root)
self.models_dir = str(models_root)
self.folder_names_and_paths = {
key: ([str(path) for path in paths], set()) for key, paths in registry.items()
}
def get_system_user_directory(self, name):
return str(self.user_root / name)
def get_folder_paths(self, name):
return list(self.folder_names_and_paths[name][0])
@pytest.mark.unit
def test_settings_live_under_comfyui_system_user_directory_and_round_trip(tmp_path):
fake = FakeFolderPaths(tmp_path / "user", tmp_path / "models", {})
expected = tmp_path / "user" / "tts_audio_suite" / "audio_cpp" / "settings.json"
settings = AudioCppSettings(
connection_mode="external",
external_server_url="http://127.0.0.1:19090",
model_roots=(str(tmp_path / "shared"),),
runtime_backend="cuda",
extras={"future_key": {"kept": True}},
)
assert get_settings_path(fake) == expected
assert save_settings(settings, folder_paths_module=fake) == expected
loaded = load_settings(folder_paths_module=fake, strict=True)
assert loaded == settings
assert json.loads(expected.read_text(encoding="utf-8"))["future_key"] == {"kept": True}
assert not list(expected.parent.glob("*.tmp"))
@pytest.mark.unit
def test_broken_settings_fail_safe_unless_strict(tmp_path):
path = tmp_path / "settings.json"
path.write_text("{broken", encoding="utf-8")
assert load_settings(path) == AudioCppSettings()
with pytest.raises(json.JSONDecodeError):
load_settings(path, strict=True)
@pytest.mark.unit
def test_model_root_precedence_deduplicates_and_keeps_managed_last(tmp_path):
explicit = tmp_path / "explicit"
configured = tmp_path / "configured"
dedicated = tmp_path / "dedicated"
tts_primary = tmp_path / "tts-primary"
tts_secondary = tmp_path / "tts-secondary"
managed = tts_primary / "audio.cpp" / "models"
fake = FakeFolderPaths(
tmp_path / "user",
tmp_path / "models",
{"audio_cpp": [dedicated], "TTS": [tts_primary, tts_secondary]},
)
settings = AudioCppSettings(
model_roots=(str(configured), str(dedicated)),
managed_model_root=str(managed),
)
assert resolve_model_roots(
[explicit, managed], settings=settings, folder_paths_module=fake
) == [
explicit,
configured,
dedicated,
tts_secondary / "audio.cpp" / "models",
managed,
]
@pytest.mark.unit
def test_discovery_prefers_existing_external_model_without_copying(tmp_path):
package = load_catalog().package("chatterbox_q8_0")
external = tmp_path / "existing-audio-cpp-models"
managed = tmp_path / "managed"
installed = package_install_path(package, external)
installed.mkdir(parents=True)
(installed / package.local_files[0]).write_bytes(b"gguf")
assert find_installed_package(package, [external, managed]) == installed
resolved = resolve_model(package.id, [external, managed])
assert resolved is not None
assert resolved.root == external
assert resolved.path == installed
assert not managed.exists()
+504
View File
@@ -0,0 +1,504 @@
import base64
import io
import json
import os
import struct
import sys
import threading
import types
import urllib.request
import wave
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from urllib.parse import parse_qs, urlsplit
import pytest
import torch
REPO_ROOT = Path(__file__).resolve().parents[2]
if str(REPO_ROOT) not in sys.path:
sys.path.insert(0, str(REPO_ROOT))
from utils.audio_cpp.client import (
AudioCppClient,
AudioCppHTTPError,
AudioCppProtocolError,
)
from utils.audio_cpp.catalog import load_catalog
from utils.audio_cpp.discovery import package_install_path
from utils.audio_cpp.process import AudioCppServerProcess, normalize_audio_cpp_task
from utils.audio_cpp.settings import AudioCppSettings
from utils.audio_cpp import resolver as audio_cpp_resolver
from utils.audio_cpp.session import (
audio_cpp_session_statuses,
close_all_audio_cpp_sessions,
get_audio_cpp_session,
)
def _wav_bytes(sample_rate=16000, channels=1, frames=32):
buffer = io.BytesIO()
with wave.open(buffer, "wb") as wav_file:
wav_file.setnchannels(channels)
wav_file.setsampwidth(2)
wav_file.setframerate(sample_rate)
samples = []
for index in range(frames):
value = int(16000 * ((index % 4) - 1.5) / 1.5)
samples.extend([value] * channels)
wav_file.writeframes(struct.pack(f"<{len(samples)}h", *samples))
return buffer.getvalue()
class _FakeAudioCppHandler(BaseHTTPRequestHandler):
server_version = "FakeAudioCpp/0.5.1"
def log_message(self, format, *args):
return None
def _json(self, payload, status=200):
body = json.dumps(payload).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_GET(self):
parsed = urlsplit(self.path)
if parsed.path == "/health":
self._json({"status": "ok", "models": 1, "features": ["unload_models"]})
elif parsed.path == "/v1/models":
self.server.model_queries += 1
self._json({
"object": "list",
"data": [{
"id": self.server.model_id,
"object": "model",
"family": "pocket_tts",
"task": "tts",
"mode": "offline",
}],
})
elif parsed.path == "/v1/audio/voices":
self.server.last_voice_query = parse_qs(parsed.query)
self._json({"voices": ["alba", "cosette"]})
else:
self._json({"error": {"message": "not found", "type": "not_found"}}, 404)
def do_POST(self):
length = int(self.headers.get("Content-Length", "0"))
payload = json.loads(self.rfile.read(length).decode("utf-8"))
self.server.requests.append(payload)
request = payload.get("request", {})
text = request.get("text")
if text == "http-error":
self._json(
{"error": {"message": "model is busy", "type": "server_busy"}},
503,
)
return
encoded = base64.b64encode(self.server.wav_bytes).decode("ascii")
if text == "named-only":
self._json({
"named_audio_outputs": [{
"id": "speech",
"audio": encoded,
"sample_rate": 16000,
"channels": 1,
}],
"timing": {},
})
elif text == "ambiguous":
self._json({
"named_audio_outputs": [
{"id": "left", "audio": encoded},
{"id": "right", "audio": encoded},
]
})
elif text == "transcript-only":
self._json({
"text": "hello world",
"language": "en",
"words": [
{"word": "hello", "start_sample": 0, "end_sample": 8000},
{"word": "world", "start_sample": 8000, "end_sample": 16000},
],
"timing": {"wall_ms": 1.0},
})
else:
self._json({
"audio": encoded,
"sample_rate": 16000,
"channels": 1,
"timing": {"wall_ms": 1.0},
})
@pytest.fixture
def fake_audio_cpp_server():
server = ThreadingHTTPServer(("127.0.0.1", 0), _FakeAudioCppHandler)
server.model_id = "pocket"
server.wav_bytes = _wav_bytes()
server.requests = []
server.last_voice_query = None
server.model_queries = 0
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
yield server, f"http://127.0.0.1:{server.server_port}"
finally:
server.shutdown()
server.server_close()
thread.join(timeout=2)
@pytest.fixture(autouse=True)
def clean_audio_cpp_sessions():
close_all_audio_cpp_sessions()
yield
close_all_audio_cpp_sessions()
@pytest.mark.unit
def test_client_health_models_voices_and_primary_audio(fake_audio_cpp_server):
server, url = fake_audio_cpp_server
client = AudioCppClient(url)
assert client.health()["status"] == "ok"
assert client.models()[0]["id"] == "pocket"
assert client.voices("pocket") == ["alba", "cosette"]
assert server.last_voice_query == {"model": ["pocket"]}
assert client.supports_feature("unload_models") is True
result = client.run_task("pocket", {"text": "hello", "seed": "42"})
assert result.sample_rate == 16000
assert result.channels == 1
assert result.waveform.shape == (1, 32)
assert result.waveform.dtype == torch.float32
assert result.waveform.device.type == "cpu"
assert server.requests[-1] == {
"model": "pocket",
"request": {"text": "hello", "seed": "42"},
}
@pytest.mark.unit
def test_client_selects_sole_named_audio_and_rejects_ambiguous_output(fake_audio_cpp_server):
_, url = fake_audio_cpp_server
client = AudioCppClient(url)
result = client.run_task("pocket", {"text": "named-only"})
assert result.sample_rate == 16000
assert list(result.named_audio) == ["speech"]
assert result.waveform.data_ptr() == result.named_audio["speech"].waveform.data_ptr()
with pytest.raises(AudioCppProtocolError, match="multiple named audio"):
client.run_task("pocket", {"text": "ambiguous"})
@pytest.mark.unit
def test_client_accepts_structured_transcript_without_audio(fake_audio_cpp_server):
_, url = fake_audio_cpp_server
result = AudioCppClient(url).run_task("asr", {"text": "transcript-only"})
assert result.waveform is None
assert result.sample_rate is None
assert result.raw["text"] == "hello world"
assert len(result.raw["words"]) == 2
@pytest.mark.unit
def test_client_surfaces_structured_http_error(fake_audio_cpp_server):
_, url = fake_audio_cpp_server
client = AudioCppClient(url)
with pytest.raises(AudioCppHTTPError) as captured:
client.run_task("pocket", {"text": "http-error"})
assert captured.value.status == 503
assert captured.value.error_type == "server_busy"
assert "model is busy" in str(captured.value)
def _write_fake_server_script(path: Path) -> Path:
script = r'''
import argparse
import base64
import io
import json
import struct
import wave
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from urllib.parse import urlsplit
parser = argparse.ArgumentParser()
parser.add_argument("--config", required=True)
args = parser.parse_args()
with open(args.config, "r", encoding="utf-8") as handle:
config = json.load(handle)
model_id = config["models"][0]["id"]
buffer = io.BytesIO()
with wave.open(buffer, "wb") as wav_file:
wav_file.setnchannels(1)
wav_file.setsampwidth(2)
wav_file.setframerate(22050)
wav_file.writeframes(struct.pack("<16h", *range(16)))
encoded = base64.b64encode(buffer.getvalue()).decode("ascii")
class Handler(BaseHTTPRequestHandler):
def log_message(self, format, *args):
return None
def send_json(self, payload):
body = json.dumps(payload).encode("utf-8")
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_GET(self):
path = urlsplit(self.path).path
if path == "/health":
self.send_json({"status": "ok", "models": 1})
elif path == "/v1/models":
self.send_json({"object": "list", "data": [{"id": model_id}]})
elif path == "/v1/audio/voices":
self.send_json({"voices": ["managed"]})
else:
self.send_json({})
def do_POST(self):
length = int(self.headers.get("Content-Length", "0"))
json.loads(self.rfile.read(length).decode("utf-8"))
self.send_json({"audio": encoded, "sample_rate": 22050, "channels": 1})
server = ThreadingHTTPServer((config["host"], config["port"]), Handler)
print("fake audio.cpp ready", flush=True)
server.serve_forever()
'''
path.write_text(script, encoding="utf-8")
return path
def _owned_config(tmp_path: Path, script: Path):
model_path = tmp_path / "model"
model_path.mkdir(exist_ok=True)
(model_path / "weights.gguf").write_bytes(b"audio-cpp-test-model")
return {
"connection_mode": "existing_binary",
"binary_path": sys.executable,
"binary_args": [str(script)],
"model_path": str(model_path),
"model_id": "managed-pocket",
"family": "pocket_tts",
"package_id": "pocket_tts_english_q8_0",
"task": "tts",
"backend": "cpu",
"startup_timeout": 5.0,
"connect_timeout": 1.0,
"request_timeout": 5.0,
"stop_timeout": 2.0,
}
@pytest.mark.unit
def test_owned_process_writes_secure_config_and_stops_exact_child(tmp_path):
assert normalize_audio_cpp_task("clone") == "clon"
assert normalize_audio_cpp_task("voice design") == "vdes"
script = _write_fake_server_script(tmp_path / "fake_audio_cpp_server.py")
runtime = AudioCppServerProcess(_owned_config(tmp_path, script)).start()
process = runtime.process
config_path = runtime.config_path
assert process is not None and process.poll() is None
assert config_path is not None and config_path.exists()
payload = json.loads(config_path.read_text(encoding="utf-8"))
assert payload["host"] == "127.0.0.1"
assert payload["cors_origins"] == ""
assert payload["log_request_body"] is False
assert payload["model_spec_override"].endswith("utils\\audio_cpp\\model_specs") or payload[
"model_spec_override"
].endswith("utils/audio_cpp/model_specs")
assert payload["models"][0]["task"] == "tts"
assert Path(payload["models"][0]["path"]).is_absolute()
assert runtime.client.voices("managed-pocket") == ["managed"]
runtime.close()
assert process.poll() is not None
assert not config_path.exists()
@pytest.mark.unit
def test_external_session_is_keyed_and_never_terminates_server(fake_audio_cpp_server):
server, url = fake_audio_cpp_server
config = {
"connection_mode": "existing_server",
"server_url": url,
"model_id": "pocket",
}
first = get_audio_cpp_session(config)
second = get_audio_cpp_session(dict(config))
assert first is second
assert first.owned is False
assert first.model_id == "pocket"
assert first.model_metadata["family"] == "pocket_tts"
assert first.task == "tts"
assert first.voices() == ["alba", "cosette"]
assert server.model_queries == 1
first.close()
with urllib.request.urlopen(f"{url}/health", timeout=1) as response:
assert json.load(response)["status"] == "ok"
@pytest.mark.unit
def test_owned_session_restarts_and_reregisters_after_exact_child_exit(tmp_path, monkeypatch):
script = _write_fake_server_script(tmp_path / "fake_audio_cpp_server.py")
fake_management = types.ModuleType("comfy.model_management")
fake_management.current_loaded_models = []
fake_management.cleanup_models = lambda: None
class LoadedModel:
def __init__(self, model):
self.model = model
fake_management.LoadedModel = LoadedModel
monkeypatch.setitem(sys.modules, "comfy.model_management", fake_management)
comfy_module = sys.modules.get("comfy")
if comfy_module is not None:
monkeypatch.setattr(comfy_module, "model_management", fake_management, raising=False)
session = get_audio_cpp_session(_owned_config(tmp_path, script))
assert session.proxy.model_size() >= len(b"audio-cpp-test-model")
first = session.run({"text": "first"})
first_runtime = session.process
first_process = first_runtime.process
assert first.sample_rate == 22050
assert len(fake_management.current_loaded_models) == 1
first_process.kill()
first_process.wait(timeout=2)
second = session.run({"text": "second"})
second_runtime = session.process
second_process = second_runtime.process
assert second.sample_rate == 22050
assert second_runtime is not first_runtime
assert second_process is not first_process
assert len(fake_management.current_loaded_models) == 1
session.restart_owned_runtime()
reset_process = session.process.process
assert second_process.poll() is not None
assert reset_process is not second_process
assert session.run({"text": "after explicit reset"}).sample_rate == 22050
assert len(fake_management.current_loaded_models) == 1
assert session.proxy.partially_unload("cpu", 1) == 0
tracked_model = fake_management.current_loaded_models[0]
session.proxy.unpatch_model("cpu")
assert reset_process.poll() is not None
assert fake_management.current_loaded_models == [tracked_model]
fake_management.current_loaded_models.pop(0)
session.run({"text": "third"})
third_process = session.process.process
assert third_process.poll() is None
assert len(fake_management.current_loaded_models) == 1
session.close()
assert third_process.poll() is not None
assert len(fake_management.current_loaded_models) == 0
@pytest.mark.unit
def test_session_makes_voice_reference_path_absolute(fake_audio_cpp_server, tmp_path, monkeypatch):
server, url = fake_audio_cpp_server
monkeypatch.chdir(tmp_path)
reference = tmp_path / "voice.wav"
reference.write_bytes(_wav_bytes())
session = get_audio_cpp_session({
"connection_mode": "existing_server",
"server_url": url,
"model_id": "pocket",
})
session.run({"text": "absolute path", "voice_ref": "voice.wav"})
sent_path = server.requests[-1]["request"]["voice_ref"]
assert Path(sent_path).is_absolute()
assert Path(sent_path) == reference
@pytest.mark.unit
def test_session_resolver_discovers_existing_model_root(tmp_path, monkeypatch):
script = _write_fake_server_script(tmp_path / "fake_audio_cpp_server.py")
external_root = tmp_path / "existing-models"
managed_root = tmp_path / "suite-managed-models"
package = load_catalog().package("pocket_tts_english_q8_0")
installed = package_install_path(package, external_root)
for relative_path in package.local_files:
target = installed / relative_path
target.parent.mkdir(parents=True, exist_ok=True)
target.write_bytes(b"installed")
monkeypatch.setattr(
audio_cpp_resolver,
"load_settings",
lambda: AudioCppSettings(
connection_mode="managed",
model_roots=(str(external_root),),
managed_model_root=str(managed_root),
runtime_backend="cpu",
),
)
session = get_audio_cpp_session({
"connection_mode": "managed",
"family": "pocket_tts",
"package_id": package.id,
"task": "tts",
"backend": "cpu",
"binary_path": sys.executable,
"binary_args": [str(script)],
"startup_timeout": 5.0,
"connect_timeout": 1.0,
"request_timeout": 5.0,
})
result = session.run({"text": "resolved"})
assert result.sample_rate == 22050
assert session.config["model_path"] == str(installed.resolve())
assert not managed_root.exists()
@pytest.mark.unit
def test_external_session_selects_the_servers_sole_model(fake_audio_cpp_server):
server, url = fake_audio_cpp_server
session = get_audio_cpp_session({
"connection_mode": "external_server",
"server_url": url,
"model_id": "",
})
reused = get_audio_cpp_session({
"connection_mode": "external_server",
"server_url": url,
"model_id": "",
})
assert reused is session
assert session.model_id == "pocket"
assert session.model_metadata["family"] == "pocket_tts"
assert session.task == "tts"
assert server.model_queries == 1
before = next(item for item in audio_cpp_session_statuses() if item["model_id"] == "pocket")
assert before["state"] == "server_ready"
assert before["owned"] is False
assert before["endpoint"] == url
session.run({"text": "status"})
after = next(item for item in audio_cpp_session_statuses() if item["model_id"] == "pocket")
assert after["state"] == "model_ready"
+1
View File
@@ -11,6 +11,7 @@ _ADAPTER_MAP: Dict[str, str] = {
"qwen3_tts": "engines.adapters.asr_qwen3_adapter.Qwen3ASREngineAdapter",
"qwen3": "engines.adapters.asr_qwen3_adapter.Qwen3ASREngineAdapter",
"granite_asr": "engines.adapters.asr_granite_adapter.GraniteASREngineAdapter",
"audio_cpp": "engines.adapters.asr_audio_cpp_adapter.AudioCppASREngineAdapter",
}
+38
View File
@@ -702,6 +702,43 @@ class OmniVoiceCacheKeyGenerator(CacheKeyGenerator):
return hashlib.md5(cache_string.encode()).hexdigest()
class AudioCppCacheKeyGenerator(CacheKeyGenerator):
"""Cache key generator for the generic audio.cpp server backend."""
def generate_cache_key(self, **params) -> str:
cache_data = {
'text': params.get('text', ''),
'audio_component': params.get('audio_component', ''),
'reference_text': params.get('reference_text', ''),
'family': params.get('family', ''),
'package_id': params.get('package_id', ''),
'model_path': params.get('model_path', ''),
'model_id': params.get('model_id', ''),
'task': params.get('task', ''),
'connection_mode': params.get('connection_mode', ''),
'server_url': params.get('server_url', ''),
'binary_path': params.get('binary_path', ''),
'backend': params.get('backend', ''),
'device_index': params.get('device_index', 0),
'language': params.get('language', ''),
'voice_id': params.get('voice_id', ''),
'instruct': params.get('instruct', ''),
'temperature': params.get('temperature'),
'top_p': params.get('top_p'),
'top_k': params.get('top_k'),
'repetition_penalty': params.get('repetition_penalty'),
'max_tokens': params.get('max_tokens'),
'max_steps': params.get('max_steps'),
'num_inference_steps': params.get('num_inference_steps'),
'guidance_scale': params.get('guidance_scale'),
'seed': params.get('seed', 0),
'request_options': params.get('request_options', ''),
'character': params.get('character', 'narrator'),
'engine': 'audio_cpp',
}
return hashlib.md5(str(sorted(cache_data.items())).encode()).hexdigest()
class AudioCache:
"""Unified audio cache manager for all TTS engines."""
@@ -720,6 +757,7 @@ class AudioCache:
'dots_tts': DotsTTSCacheKeyGenerator(),
'dramabox': DramaBoxCacheKeyGenerator(),
'fish_audio_s2': FishAudioS2CacheKeyGenerator(),
'audio_cpp': AudioCppCacheKeyGenerator(),
'omnivoice': OmniVoiceCacheKeyGenerator(),
'moss_tts': MossTTSCacheKeyGenerator(),
'moss_soundeffect_v2': MossSoundEffectV2CacheKeyGenerator(),
+43
View File
@@ -0,0 +1,43 @@
"""Pinned audio.cpp integration data, discovery, and installation helpers."""
from .catalog import (
AUDIO_CPP_RELEASE_COMMIT,
AUDIO_CPP_RELEASE_TAG,
AUDIO_CPP_RELEASE_VERSION,
AudioCppCatalog,
FamilyRecord,
PackageRecord,
family_choices,
get_family,
get_model_specs_dir,
get_package,
load_catalog,
package_choices,
recommended_package,
resolve_task,
)
from .settings import AudioCppSettings, get_settings_path, load_settings, save_settings
from .resolver import AudioCppResolutionError, resolve_audio_cpp_config
__all__ = [
"AUDIO_CPP_RELEASE_COMMIT",
"AUDIO_CPP_RELEASE_TAG",
"AUDIO_CPP_RELEASE_VERSION",
"AudioCppCatalog",
"AudioCppSettings",
"AudioCppResolutionError",
"FamilyRecord",
"PackageRecord",
"family_choices",
"get_family",
"get_model_specs_dir",
"get_package",
"get_settings_path",
"load_catalog",
"load_settings",
"package_choices",
"recommended_package",
"resolve_audio_cpp_config",
"resolve_task",
"save_settings",
]
+184
View File
@@ -0,0 +1,184 @@
"""Resolve Suite integration capabilities for pinned audio.cpp families."""
from __future__ import annotations
from copy import deepcopy
from functools import lru_cache
from pathlib import Path
from typing import Any, Dict, Mapping
import yaml
from .catalog import AUDIO_CPP_RELEASE_TAG, PackageRecord, load_catalog
class CapabilityError(ValueError):
pass
def get_capability_path() -> Path:
return Path(__file__).resolve().with_name("integration_capabilities.yaml")
@lru_cache(maxsize=1)
def _load_overlay() -> Dict[str, Any]:
path = get_capability_path()
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
except (OSError, yaml.YAMLError) as exc:
raise CapabilityError(f"Cannot read audio.cpp capability overlay {path}: {exc}") from exc
if raw.get("schema_version") != 1 or raw.get("release") != AUDIO_CPP_RELEASE_TAG:
raise CapabilityError("audio.cpp capability overlay release/schema does not match the catalog")
return dict(raw)
def _merge(base: Mapping[str, Any], override: Mapping[str, Any]) -> Dict[str, Any]:
value = deepcopy(dict(base))
for key, item in override.items():
if isinstance(item, Mapping) and isinstance(value.get(key), Mapping):
value[key] = _merge(value[key], item)
else:
value[key] = deepcopy(item)
return value
@lru_cache(maxsize=1)
def load_capabilities() -> Dict[str, Dict[str, Any]]:
raw = _load_overlay()
catalog = load_catalog()
families = raw.get("families") or {}
if set(families) != set(catalog.families):
missing = sorted(set(catalog.families) - set(families))
extra = sorted(set(families) - set(catalog.families))
raise CapabilityError(f"audio.cpp capability families mismatch; missing={missing}, extra={extra}")
resolved: Dict[str, Dict[str, Any]] = {}
defaults = raw.get("defaults") or {}
for family_id, override in families.items():
family = catalog.family(family_id)
item = _merge(defaults, override or {})
item.update(
id=family.id,
display_name=family.display_name,
description=family.description,
languages=list(family.languages),
upstream_tasks=list(family.runtime_tasks),
packages=[package.id for package in family.packages],
recommended_package_id=family.recommended_package_id,
)
asr_features = item.get("asr_features") or {}
if not isinstance(asr_features, Mapping):
raise CapabilityError(
f"audio.cpp {family_id} asr_features must be an object"
)
diarization = str(asr_features.get("diarization", "none"))
timing = str(asr_features.get("timing", "none"))
if diarization not in {"none", "native"}:
raise CapabilityError(
f"audio.cpp {family_id} has invalid ASR diarization capability: {diarization}"
)
if timing not in {
"none",
"native_word",
"native_segment",
"optional_forced_aligner",
}:
raise CapabilityError(
f"audio.cpp {family_id} has invalid ASR timing capability: {timing}"
)
resolved[family_id] = item
return resolved
def get_capability(family: str) -> Dict[str, Any]:
try:
return deepcopy(load_capabilities()[str(family)])
except KeyError as exc:
raise CapabilityError(f"Unknown audio.cpp capability family: {family!r}") from exc
def public_capabilities() -> Dict[str, Any]:
catalog = load_catalog()
raw_sizes = _load_overlay().get("package_sizes") or {}
if set(raw_sizes) != set(catalog.packages):
missing = sorted(set(catalog.packages) - set(raw_sizes))
extra = sorted(set(raw_sizes) - set(catalog.packages))
raise CapabilityError(f"audio.cpp package sizes mismatch; missing={missing}, extra={extra}")
packages = {}
for package_id, package in catalog.packages.items():
size = raw_sizes[package_id]
if not isinstance(size, int) or size <= 0:
raise CapabilityError(f"Invalid estimated size for audio.cpp package {package_id!r}")
dependencies = get_package_dependencies(package_id)
dependency_bytes = sum(
int(item["estimated_download_bytes"]) for item in dependencies
)
packages[package_id] = {
"id": package.id,
"family": package.family,
"display_name": package.display_name,
"format": package.format,
"precision": package.precision,
"estimated_download_bytes": size + dependency_bytes,
"primary_download_bytes": size,
"dependencies": [item["package"].id for item in dependencies],
}
return {
"schema_version": 1,
"release": AUDIO_CPP_RELEASE_TAG,
"sizes_checked_at": "2026-08-13",
"families": load_capabilities(),
"packages": packages,
}
def get_package_dependencies(package_id: str) -> list[Dict[str, Any]]:
entries = (_load_overlay().get("package_dependencies") or {}).get(package_id, [])
dependencies = []
for entry in entries:
package = PackageRecord(
family=str(entry["family"]),
id=str(entry["id"]),
display_name=str(entry["display_name"]),
target_directory=str(entry["target_directory"]),
format=str(entry["format"]),
precision=str(entry["precision"]),
files=tuple(str(value) for value in entry["files"]),
strip_prefix=str(entry.get("strip_prefix", "")),
download={
"kind": "huggingface_snapshot",
"repo": str(entry["repo"]),
"revision": str(entry.get("revision", "main")),
"gated": False,
},
)
# Validate local mappings before the downloader touches disk.
package.local_files
dependencies.append(
{
"package": package,
"session_option": str(entry["session_option"]),
"estimated_download_bytes": int(entry["estimated_download_bytes"]),
}
)
return dependencies
def validate_voice_reference(family: str, voice_ref: Any, character: str = "narrator") -> None:
"""Enforce requirements that the frontend panel merely explains."""
from utils.voice.reference import effective_voice_audio
capability = get_capability(family)
audio_requirement = capability["reference_audio"]
has_audio = isinstance(voice_ref, Mapping) and effective_voice_audio(voice_ref) is not None
if audio_requirement in {"required", "required_per_speaker"} and not has_audio:
raise ValueError(f"audio.cpp {family} requires reference audio for '{character}'")
transcript = ""
if isinstance(voice_ref, Mapping):
transcript = str(
voice_ref.get("reference_text") or voice_ref.get("prompt_text") or voice_ref.get("text") or ""
).strip()
if capability["reference_transcript"] == "required" and not transcript:
raise ValueError(
f"audio.cpp {family} requires the transcript matching '{character}' reference audio"
)
+38
View File
@@ -0,0 +1,38 @@
"""HTTP route for the audio.cpp engine capability panel."""
from __future__ import annotations
def register_audio_cpp_capability_routes(routes, web) -> None:
@routes.get("/api/tts-audio-suite/audio-cpp-capabilities")
async def get_audio_cpp_capabilities(_request):
try:
from .capabilities import public_capabilities
return web.json_response(public_capabilities())
except Exception as exc:
return web.json_response({"error": str(exc)}, status=500)
@routes.get("/api/tts-audio-suite/audio-cpp-status")
async def get_audio_cpp_status(_request):
try:
from .session import audio_cpp_session_statuses
return web.json_response({"sessions": audio_cpp_session_statuses()})
except Exception as exc:
return web.json_response({"sessions": [], "error": str(exc)}, status=500)
@routes.post("/api/tts-audio-suite/audio-cpp-stop")
async def stop_audio_cpp_session(request):
try:
from .session import stop_owned_audio_cpp_session
data = await request.json()
stopped = stop_owned_audio_cpp_session(str(data.get("session_id", "")))
if not stopped:
return web.json_response({"error": "audio.cpp session not found"}, status=404)
return web.json_response({"status": "stopped"})
except PermissionError as exc:
return web.json_response({"error": str(exc)}, status=403)
except Exception as exc:
return web.json_response({"error": str(exc)}, status=500)
+368
View File
@@ -0,0 +1,368 @@
"""Pinned audio.cpp release-0.5.1 Suite-compatible model catalog.
The bundled JSON files are exact copies of the selected upstream tag's model
specifications. The executable's compiled task IDs are kept separately because
several release specs expose broader or differently-spelled task metadata.
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from functools import lru_cache
from pathlib import Path, PurePosixPath
from typing import Any, Dict, Iterator, Mapping, Optional, Sequence, Tuple
AUDIO_CPP_RELEASE_VERSION = "0.5.1"
AUDIO_CPP_RELEASE_TAG = "release-0.5.1"
AUDIO_CPP_RELEASE_COMMIT = "238ab6a9e321c17de8e120559f57efeedaeb1345"
MODEL_SPEC_FILENAMES: Tuple[str, ...] = (
"citrinet_asr.json",
"chatterbox.json",
"confucius4_tts.json",
"dramabox.json",
"fish_audio.json",
"fun_asr_nano.json",
"glm_tts.json",
"higgs_audio_tts.json",
"higgs_audio_stt.json",
"hviske_asr.json",
"index_tts2.json",
"inflect_v2.json",
"irodori_tts.json",
"kroko_asr.json",
"miotts.json",
"moss_tts_local.json",
"moss_tts_nano.json",
"nemotron_asr.json",
"omnivoice.json",
"outetts.json",
"parakeet_tdt.json",
"pocket_tts.json",
"qwen3_tts.json",
"qwen3_asr.json",
"seed_vc.json",
"supertonic.json",
"vevo2.json",
"vibevoice.json",
"vibevoice_asr.json",
"vietneu_tts.json",
"voxcpm2.json",
"voxtral_realtime.json",
)
# These are the task IDs actually compiled into release-0.5.1. Do not derive
# them from the broader human-facing ``tasks`` arrays in the JSON specs.
COMPILED_TASKS: Mapping[str, Tuple[str, ...]] = {
"citrinet_asr": ("asr",),
"chatterbox": ("clon", "vc"),
"confucius4_tts": ("clon",),
"dramabox": ("tts", "clon"),
"fish_audio": ("tts",),
"fun_asr_nano": ("asr",),
"glm_tts": ("tts", "clon"),
"higgs_audio_tts": ("tts",),
"higgs_audio_stt": ("asr",),
"hviske_asr": ("asr",),
"index_tts2": ("tts", "clon"),
"inflect_v2": ("tts",),
"irodori_tts": ("tts", "clon", "vdes"),
"kroko_asr": ("asr",),
"miotts": ("tts",),
"moss_tts_local": ("tts", "clon"),
"moss_tts_nano": ("tts", "clon"),
"nemotron_asr": ("asr",),
"omnivoice": ("tts",),
"outetts": ("tts", "clon"),
"parakeet_tdt": ("asr",),
"pocket_tts": ("tts",),
"qwen3_tts": ("tts", "vdes"),
"qwen3_asr": ("asr",),
"seed_vc": ("vc", "svc"),
"supertonic": ("tts",),
"vevo2": ("tts", "vc", "s2s", "svc"),
"vibevoice": ("tts",),
"vibevoice_asr": ("asr",),
"vietneu_tts": ("tts", "vdes"),
"voxcpm2": ("tts",),
"voxtral_realtime": ("asr",),
}
_TASK_ALIASES = {
"clone": "clon",
"cloning": "clon",
"voice_clone": "clon",
"voice_cloning": "clon",
"voice_design": "vdes",
"design": "vdes",
"voice_conversion": "vc",
"speech_to_speech": "s2s",
"singing_voice_conversion": "svc",
}
class CatalogError(ValueError):
"""Raised when a pinned model specification is missing or inconsistent."""
def _safe_package_relative_path(value: str, label: str) -> PurePosixPath:
normalized = value.replace("\\", "/")
path = PurePosixPath(normalized)
if (
not normalized
or path.is_absolute()
or ".." in path.parts
or (path.parts and ":" in path.parts[0])
):
raise CatalogError(f"Unsafe {label}: {value!r}")
return path
@dataclass(frozen=True)
class PackageRecord:
family: str
id: str
display_name: str
target_directory: str
format: str
precision: str
files: Tuple[str, ...]
strip_prefix: str
download: Mapping[str, Any]
default: bool = False
def local_relative_path(self, remote_path: str) -> Path:
"""Map one remote package path to its installed relative path."""
remote = _safe_package_relative_path(remote_path, "remote file path")
prefix_text = self.strip_prefix.replace("\\", "/").rstrip("/")
if prefix_text in ("", "."):
local = remote
else:
prefix = _safe_package_relative_path(prefix_text, "strip_prefix")
if remote == prefix or remote.parts[: len(prefix.parts)] != prefix.parts:
raise CatalogError(
f"Package {self.id!r} file {remote_path!r} is outside "
f"strip_prefix {self.strip_prefix!r}"
)
local = PurePosixPath(*remote.parts[len(prefix.parts) :])
if not local.parts:
raise CatalogError(f"Package {self.id!r} maps {remote_path!r} to an empty path")
return Path(*local.parts)
@property
def local_files(self) -> Tuple[Path, ...]:
return tuple(self.local_relative_path(path) for path in self.files)
@property
def repo(self) -> str:
return str(self.download.get("repo", ""))
@property
def revision(self) -> str:
return str(self.download.get("revision", "main"))
@property
def gated(self) -> bool:
return bool(self.download.get("gated", False))
@dataclass(frozen=True)
class FamilyRecord:
id: str
display_name: str
description: str
category: str
status: str
runtime_tasks: Tuple[str, ...]
languages: Tuple[str, ...]
options: Mapping[str, Any]
packages: Tuple[PackageRecord, ...]
recommended_package_id: str
spec_filename: str
raw: Mapping[str, Any]
@dataclass(frozen=True)
class AudioCppCatalog:
families: Mapping[str, FamilyRecord]
packages: Mapping[str, PackageRecord]
specs_dir: Path
release_version: str = AUDIO_CPP_RELEASE_VERSION
def iter_families(self) -> Iterator[FamilyRecord]:
return iter(self.families.values())
def iter_packages(self, family: Optional[str] = None) -> Iterator[PackageRecord]:
if family is None:
return iter(self.packages.values())
return iter(self.family(family).packages)
def family(self, family_id: str) -> FamilyRecord:
try:
return self.families[family_id]
except KeyError as exc:
raise CatalogError(f"Unknown audio.cpp family: {family_id!r}") from exc
def package(self, package_id: str) -> PackageRecord:
try:
return self.packages[package_id]
except KeyError as exc:
raise CatalogError(f"Unknown audio.cpp package: {package_id!r}") from exc
def get_model_specs_dir() -> Path:
"""Return the exact release-0.5.1 spec directory shipped with the node."""
return Path(__file__).resolve().parent / "model_specs"
def _merged_download(spec: Mapping[str, Any], package: Mapping[str, Any]) -> Dict[str, Any]:
merged = dict(spec.get("package_defaults", {}).get("download", {}))
merged.update(package.get("download", {}))
return merged
def _load_catalog(specs_dir: Path) -> AudioCppCatalog:
families: Dict[str, FamilyRecord] = {}
packages: Dict[str, PackageRecord] = {}
for filename in MODEL_SPEC_FILENAMES:
path = specs_dir / filename
if not path.is_file():
raise CatalogError(f"Missing pinned audio.cpp model spec: {path}")
try:
raw = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
raise CatalogError(f"Cannot read audio.cpp model spec {path}: {exc}") from exc
family_id = str(raw.get("family", ""))
if family_id not in COMPILED_TASKS:
raise CatalogError(f"Spec {filename} has unsupported release family {family_id!r}")
if family_id in families:
raise CatalogError(f"Duplicate audio.cpp family {family_id!r}")
family_packages = []
for package_data in raw.get("packages", []):
package_id = str(package_data.get("id", ""))
if not package_id or package_id in packages:
raise CatalogError(f"Missing or duplicate package ID {package_id!r} in {filename}")
package = PackageRecord(
family=family_id,
id=package_id,
display_name=str(package_data.get("display_name", package_id)),
target_directory=str(package_data.get("target_directory", "")),
format=str(package_data.get("format", "")),
precision=str(package_data.get("precision", "")),
files=tuple(str(item) for item in package_data.get("files", [])),
strip_prefix=str(package_data.get("strip_prefix", "")),
download=_merged_download(raw, package_data),
default=bool(package_data.get("default", False)),
)
_safe_package_relative_path(package.target_directory, "target_directory")
if not package.files:
raise CatalogError(f"Package {package_id!r} has no downloadable files")
# Validate prefix mappings when loading, before any filesystem mutation.
package.local_files
if package.download.get("kind") != "huggingface_snapshot" or not package.repo:
raise CatalogError(f"Package {package_id!r} has no supported download source")
packages[package_id] = package
family_packages.append(package)
recommendation = str(raw.get("ui", {}).get("recommended_package", ""))
if not recommendation:
recommendation = next((p.id for p in family_packages if p.default), "")
if recommendation not in {package.id for package in family_packages}:
raise CatalogError(
f"Family {family_id!r} recommends unknown package {recommendation!r}"
)
families[family_id] = FamilyRecord(
id=family_id,
display_name=str(raw.get("display_name", family_id)),
description=str(raw.get("description", "")),
category=str(raw.get("category", "")),
status=str(raw.get("status", "")),
runtime_tasks=COMPILED_TASKS[family_id],
languages=tuple(str(item) for item in raw.get("languages", [])),
options=dict(raw.get("options", {})),
packages=tuple(family_packages),
recommended_package_id=recommendation,
spec_filename=filename,
raw=raw,
)
if set(families) != set(COMPILED_TASKS):
missing = sorted(set(COMPILED_TASKS) - set(families))
raise CatalogError(f"Pinned audio.cpp catalog is incomplete; missing {missing}")
if len(packages) != 96:
raise CatalogError(f"Expected 96 Suite-compatible release-0.5.1 packages, found {len(packages)}")
return AudioCppCatalog(families=families, packages=packages, specs_dir=specs_dir)
@lru_cache(maxsize=1)
def _load_bundled_catalog() -> AudioCppCatalog:
return _load_catalog(get_model_specs_dir())
def load_catalog(specs_dir: Optional[Path] = None) -> AudioCppCatalog:
"""Load the bundled catalog, or validate an equivalent override directory."""
if specs_dir is None:
return _load_bundled_catalog()
return _load_catalog(Path(specs_dir).expanduser().resolve())
def family_choices() -> list[str]:
return list(load_catalog().families)
def package_choices(family: Optional[str] = None) -> list[str]:
catalog = load_catalog()
if family is None:
return list(catalog.packages)
return [package.id for package in catalog.family(family).packages]
def get_family(family: str) -> FamilyRecord:
return load_catalog().family(family)
def get_package(package_id: str) -> PackageRecord:
return load_catalog().package(package_id)
def recommended_package(family: str) -> str:
return get_family(family).recommended_package_id
def resolve_task(family: str, package_id: Optional[str], requested: str = "auto") -> str:
"""Resolve a UI task name to a release-0.5.1 compiled task ID."""
catalog = load_catalog()
family_record = catalog.family(family)
package = catalog.package(package_id) if package_id is not None else None
if package is not None and package.family != family:
raise CatalogError(f"Package {package_id!r} does not belong to family {family!r}")
normalized = str(requested or "auto").strip().lower().replace("-", "_").replace(" ", "_")
if normalized == "auto":
if package is not None and "vdes" in family_record.runtime_tasks:
package_label = f"{package.id} {package.display_name}".lower().replace("_", "")
if "voicedesign" in package_label:
return "vdes"
return family_record.runtime_tasks[0]
normalized = _TASK_ALIASES.get(normalized, normalized)
# Several release families condition cloning through a speaker reference on
# the compiled ``tts`` task instead of exposing a separate ``clon`` task.
if normalized == "clon" and "clon" not in family_record.runtime_tasks:
if "tts" in family_record.runtime_tasks:
return "tts"
if normalized not in family_record.runtime_tasks:
supported = ", ".join(family_record.runtime_tasks)
raise CatalogError(
f"Task {requested!r} is unavailable for {family!r} in audio.cpp "
f"{AUDIO_CPP_RELEASE_VERSION}; supported: {supported}"
)
return normalized
+431
View File
@@ -0,0 +1,431 @@
"""Small stdlib HTTP client for the audio.cpp server API."""
from __future__ import annotations
import array
import base64
import binascii
import io
import json
import socket
import sys
import urllib.error
import urllib.parse
import urllib.request
import wave
from dataclasses import dataclass, field
from typing import Any, Dict, Mapping, Optional
import torch
class AudioCppClientError(RuntimeError):
"""Base error for audio.cpp transport failures."""
class AudioCppConnectionError(AudioCppClientError):
"""The audio.cpp endpoint could not be reached."""
class AudioCppTimeoutError(AudioCppClientError):
"""The audio.cpp endpoint did not respond before the configured timeout."""
class AudioCppProtocolError(AudioCppClientError):
"""The server returned a response that does not match its API contract."""
class AudioCppHTTPError(AudioCppClientError):
"""Structured non-success response from audio.cpp."""
def __init__(
self,
status: int,
message: str,
*,
error_type: Optional[str] = None,
path: str = "",
response_body: str = "",
) -> None:
self.status = int(status)
self.error_type = error_type
self.path = path
self.response_body = response_body
label = f"audio.cpp HTTP {self.status}"
if error_type:
label += f" ({error_type})"
if path:
label += f" for {path}"
super().__init__(f"{label}: {message}")
@dataclass(frozen=True)
class AudioCppAudio:
"""Decoded PCM audio returned by audio.cpp."""
waveform: torch.Tensor
sample_rate: int
channels: int
@dataclass(frozen=True)
class AudioCppTaskResult:
"""Decoded result from ``POST /v1/tasks/run``."""
waveform: Optional[torch.Tensor] = None
sample_rate: Optional[int] = None
channels: Optional[int] = None
named_audio: Dict[str, AudioCppAudio] = field(default_factory=dict)
raw: Dict[str, Any] = field(default_factory=dict)
def _decode_pcm_wav(wav_bytes: bytes, *, context: str = "audio") -> AudioCppAudio:
if not isinstance(wav_bytes, (bytes, bytearray)) or not wav_bytes:
raise AudioCppProtocolError(f"audio.cpp returned empty {context} WAV data")
try:
with wave.open(io.BytesIO(bytes(wav_bytes)), "rb") as wav_file:
channels = int(wav_file.getnchannels())
sample_rate = int(wav_file.getframerate())
sample_width = int(wav_file.getsampwidth())
frame_count = int(wav_file.getnframes())
compression = wav_file.getcomptype()
frames = wav_file.readframes(frame_count)
except (EOFError, wave.Error) as exc:
raise AudioCppProtocolError(f"audio.cpp returned an invalid {context} WAV: {exc}") from exc
if compression != "NONE":
raise AudioCppProtocolError(
f"audio.cpp returned unsupported compressed {context} WAV data ({compression})"
)
if channels <= 0 or sample_rate <= 0:
raise AudioCppProtocolError(
f"audio.cpp returned invalid {context} WAV metadata "
f"(sample_rate={sample_rate}, channels={channels})"
)
if sample_width not in (1, 2, 3, 4):
raise AudioCppProtocolError(
f"audio.cpp returned unsupported {context} WAV sample width: {sample_width} bytes"
)
expected_samples = frame_count * channels
expected_bytes = expected_samples * sample_width
if len(frames) != expected_bytes:
raise AudioCppProtocolError(
f"audio.cpp returned truncated {context} WAV data "
f"({len(frames)} bytes, expected {expected_bytes})"
)
if sample_width == 1:
values = torch.tensor(list(frames), dtype=torch.float32)
values = (values - 128.0) / 128.0
elif sample_width == 2:
pcm = array.array("h")
pcm.frombytes(frames)
if sys.byteorder != "little":
pcm.byteswap()
values = torch.tensor(pcm, dtype=torch.float32) / 32768.0
elif sample_width == 3:
decoded = []
for offset in range(0, len(frames), 3):
sample = int.from_bytes(frames[offset : offset + 3], "little", signed=False)
if sample & 0x800000:
sample -= 0x1000000
decoded.append(sample)
values = torch.tensor(decoded, dtype=torch.float32) / 8388608.0
else:
pcm = array.array("i")
pcm.frombytes(frames)
if sys.byteorder != "little":
pcm.byteswap()
values = torch.tensor(pcm, dtype=torch.float32) / 2147483648.0
if expected_samples == 0:
waveform = torch.empty((channels, 0), dtype=torch.float32)
else:
waveform = values.reshape(frame_count, channels).transpose(0, 1).contiguous()
return AudioCppAudio(
waveform=waveform.cpu(),
sample_rate=sample_rate,
channels=channels,
)
def _decode_base64_wav(value: Any, *, context: str) -> AudioCppAudio:
if not isinstance(value, str) or not value:
raise AudioCppProtocolError(f"audio.cpp response field '{context}' must be base64 WAV text")
try:
wav_bytes = base64.b64decode(value, validate=True)
except (ValueError, binascii.Error) as exc:
raise AudioCppProtocolError(
f"audio.cpp response field '{context}' is not valid base64"
) from exc
return _decode_pcm_wav(wav_bytes, context=context)
class AudioCppClient:
"""Synchronous audio.cpp HTTP client using only Python's standard library."""
def __init__(
self,
base_url: str,
*,
connect_timeout: float = 5.0,
request_timeout: float = 600.0,
max_response_bytes: int = 2 * 1024 * 1024 * 1024,
) -> None:
parsed = urllib.parse.urlsplit(str(base_url).strip())
if parsed.scheme not in ("http", "https") or not parsed.netloc:
raise ValueError(f"Invalid audio.cpp server URL: {base_url!r}")
if parsed.query or parsed.fragment:
raise ValueError("audio.cpp server URL must not contain a query or fragment")
self.base_url = urllib.parse.urlunsplit(
(parsed.scheme, parsed.netloc, parsed.path.rstrip("/"), "", "")
)
self.connect_timeout = max(0.01, float(connect_timeout))
self.request_timeout = max(0.01, float(request_timeout))
self.max_response_bytes = max(1, int(max_response_bytes))
def _url(self, path: str) -> str:
if not path.startswith("/"):
path = "/" + path
return self.base_url + path
@staticmethod
def _error_details(body: bytes, fallback: str) -> tuple[str, Optional[str], str]:
text = body.decode("utf-8", errors="replace")
message = fallback
error_type = None
try:
payload = json.loads(text)
error = payload.get("error") if isinstance(payload, dict) else None
if isinstance(error, dict):
message = str(error.get("message") or fallback)
error_type = str(error.get("type")) if error.get("type") else None
elif error:
message = str(error)
except (TypeError, ValueError):
if text.strip():
message = text.strip()[:1000]
return message, error_type, text
@staticmethod
def _is_timeout_error(exc: BaseException) -> bool:
if isinstance(exc, (TimeoutError, socket.timeout)):
return True
if isinstance(exc, urllib.error.URLError):
return isinstance(exc.reason, (TimeoutError, socket.timeout))
return False
def _request_bytes(
self,
method: str,
path: str,
*,
payload: Optional[Mapping[str, Any]] = None,
timeout: Optional[float] = None,
) -> tuple[bytes, Mapping[str, str]]:
body = None
headers = {"Accept": "application/json"}
if payload is not None:
try:
body = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
except (TypeError, ValueError) as exc:
raise ValueError(f"audio.cpp request is not JSON serializable: {exc}") from exc
headers["Content-Type"] = "application/json"
request = urllib.request.Request(
self._url(path),
data=body,
headers=headers,
method=method.upper(),
)
effective_timeout = self.request_timeout if timeout is None else max(0.01, float(timeout))
try:
with urllib.request.urlopen(request, timeout=effective_timeout) as response:
content_length = response.headers.get("Content-Length")
if content_length:
try:
if int(content_length) > self.max_response_bytes:
raise AudioCppProtocolError(
f"audio.cpp response exceeds {self.max_response_bytes} bytes"
)
except ValueError:
pass
response_body = response.read(self.max_response_bytes + 1)
if len(response_body) > self.max_response_bytes:
raise AudioCppProtocolError(
f"audio.cpp response exceeds {self.max_response_bytes} bytes"
)
return response_body, response.headers
except urllib.error.HTTPError as exc:
error_body = exc.read(65536)
message, error_type, response_text = self._error_details(error_body, str(exc.reason))
raise AudioCppHTTPError(
exc.code,
message,
error_type=error_type,
path=path,
response_body=response_text,
) from exc
except AudioCppClientError:
raise
except (urllib.error.URLError, TimeoutError, socket.timeout, OSError) as exc:
if self._is_timeout_error(exc):
raise AudioCppTimeoutError(
f"audio.cpp request to {path} timed out after {effective_timeout:.2f}s"
) from exc
reason = exc.reason if isinstance(exc, urllib.error.URLError) else exc
raise AudioCppConnectionError(
f"Could not connect to audio.cpp at {self.base_url}: {reason}"
) from exc
def _request_json(
self,
method: str,
path: str,
*,
payload: Optional[Mapping[str, Any]] = None,
timeout: Optional[float] = None,
) -> Dict[str, Any]:
response_body, _ = self._request_bytes(method, path, payload=payload, timeout=timeout)
try:
decoded = json.loads(response_body.decode("utf-8"))
except (UnicodeDecodeError, ValueError) as exc:
raise AudioCppProtocolError(
f"audio.cpp returned invalid JSON for {path}: {exc}"
) from exc
if not isinstance(decoded, dict):
raise AudioCppProtocolError(
f"audio.cpp returned {type(decoded).__name__}, expected an object for {path}"
)
return decoded
def health(self, *, timeout: Optional[float] = None) -> Dict[str, Any]:
return self._request_json(
"GET",
"/health",
timeout=self.connect_timeout if timeout is None else timeout,
)
def models(self, *, timeout: Optional[float] = None) -> list[Dict[str, Any]]:
payload = self._request_json("GET", "/v1/models", timeout=timeout)
models = payload.get("data")
if not isinstance(models, list) or any(not isinstance(item, dict) for item in models):
raise AudioCppProtocolError("audio.cpp /v1/models response is missing a valid data list")
return list(models)
def voices(self, model_id: str, *, timeout: Optional[float] = None) -> list[str]:
query = urllib.parse.urlencode({"model": str(model_id)})
payload = self._request_json("GET", f"/v1/audio/voices?{query}", timeout=timeout)
voices = payload.get("voices")
if not isinstance(voices, list) or any(not isinstance(item, str) for item in voices):
raise AudioCppProtocolError("audio.cpp voices response is missing a valid voices list")
return list(voices)
def features(self, *, timeout: Optional[float] = None) -> set[str]:
"""Read optional future feature metadata without assuming release-0.5.1 has it."""
payload = self.health(timeout=timeout)
raw = payload.get("features", payload.get("capabilities", []))
if isinstance(raw, dict):
return {str(name) for name, enabled in raw.items() if enabled}
if isinstance(raw, list):
return {str(item) for item in raw}
return set()
def supports_feature(self, name: str, *, timeout: Optional[float] = None) -> bool:
return str(name) in self.features(timeout=timeout)
def run_task(
self,
model_id: str,
request: Mapping[str, Any],
*,
timeout: Optional[float] = None,
) -> AudioCppTaskResult:
if not isinstance(request, Mapping):
raise TypeError("audio.cpp task request must be a mapping")
payload = self._request_json(
"POST",
"/v1/tasks/run",
payload={"model": str(model_id), "request": dict(request)},
timeout=timeout,
)
named_audio: Dict[str, AudioCppAudio] = {}
raw_named = payload.get("named_audio_outputs", [])
if raw_named is None:
raw_named = []
if not isinstance(raw_named, list):
raise AudioCppProtocolError("audio.cpp named_audio_outputs must be a list")
for index, item in enumerate(raw_named):
if not isinstance(item, dict):
raise AudioCppProtocolError(
f"audio.cpp named_audio_outputs[{index}] must be an object"
)
output_id = item.get("id")
if not isinstance(output_id, str) or not output_id:
raise AudioCppProtocolError(
f"audio.cpp named_audio_outputs[{index}] is missing a non-empty id"
)
if output_id in named_audio:
raise AudioCppProtocolError(f"audio.cpp returned duplicate named audio id: {output_id}")
decoded = _decode_base64_wav(
item.get("audio"), context=f"named_audio_outputs[{index}].audio"
)
self._validate_declared_audio_metadata(item, decoded, f"named_audio_outputs[{index}]")
named_audio[output_id] = decoded
primary: Optional[AudioCppAudio] = None
if "audio" in payload and payload.get("audio") is not None:
primary = _decode_base64_wav(payload.get("audio"), context="audio")
self._validate_declared_audio_metadata(payload, primary, "audio")
elif len(named_audio) == 1:
primary = next(iter(named_audio.values()))
elif len(named_audio) > 1:
raise AudioCppProtocolError(
"audio.cpp task result contains multiple named audio outputs but no primary audio"
)
elif not named_audio and not any(
key in payload for key in ("text", "segments", "speaker_turns", "words")
):
raise AudioCppProtocolError("audio.cpp task result did not contain task output")
return AudioCppTaskResult(
waveform=primary.waveform if primary is not None else None,
sample_rate=primary.sample_rate if primary is not None else None,
channels=primary.channels if primary is not None else None,
named_audio=named_audio,
raw=payload,
)
@staticmethod
def _validate_declared_audio_metadata(
payload: Mapping[str, Any],
decoded: AudioCppAudio,
context: str,
) -> None:
declared_rate = payload.get("sample_rate")
if declared_rate is not None and int(declared_rate) != decoded.sample_rate:
raise AudioCppProtocolError(
f"audio.cpp {context} sample rate metadata ({declared_rate}) "
f"does not match its WAV ({decoded.sample_rate})"
)
declared_channels = payload.get("channels")
if declared_channels is not None and int(declared_channels) != decoded.channels:
raise AudioCppProtocolError(
f"audio.cpp {context} channel metadata ({declared_channels}) "
f"does not match its WAV ({decoded.channels})"
)
__all__ = [
"AudioCppAudio",
"AudioCppClient",
"AudioCppClientError",
"AudioCppConnectionError",
"AudioCppHTTPError",
"AudioCppProtocolError",
"AudioCppTaskResult",
"AudioCppTimeoutError",
]
+187
View File
@@ -0,0 +1,187 @@
"""Resolve existing audio.cpp models without copying or cache migration."""
from __future__ import annotations
import os
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Optional, Sequence, Union
from .catalog import AudioCppCatalog, PackageRecord, load_catalog
from .settings import AudioCppSettings, get_settings_path, load_settings
def _import_folder_paths():
try:
import folder_paths # type: ignore
return folder_paths
except (ImportError, RuntimeError):
return None
def _registered_paths(folder_paths_module, key: str) -> list[Path]:
if folder_paths_module is None:
return []
registry = getattr(folder_paths_module, "folder_names_and_paths", {})
if key not in registry:
return []
try:
if hasattr(folder_paths_module, "get_folder_paths"):
values = folder_paths_module.get_folder_paths(key)
else:
values = registry[key][0]
except (KeyError, TypeError, ValueError):
return []
return [Path(value).expanduser() for value in values if value]
def _tts_paths(folder_paths_module) -> list[Path]:
paths: list[Path] = []
for key in ("TTS", "tts"):
paths.extend(_registered_paths(folder_paths_module, key))
if not paths and folder_paths_module is not None:
models_dir = getattr(folder_paths_module, "models_dir", None)
if models_dir:
paths.append(Path(models_dir) / "TTS")
return paths
def default_managed_model_root(
*,
settings: Optional[AudioCppSettings] = None,
folder_paths_module=None,
) -> Path:
current = settings or AudioCppSettings()
if current.managed_model_root:
return Path(current.managed_model_root).expanduser()
module = folder_paths_module if folder_paths_module is not None else _import_folder_paths()
tts_paths = _tts_paths(module)
if tts_paths:
return tts_paths[0] / "audio.cpp" / "models"
# This is only a non-ComfyUI safety fallback; it remains direct storage, not
# a Hugging Face cache.
return get_settings_path(module).parent / "models"
def _canonical(path: Path) -> str:
return os.path.normcase(os.path.abspath(os.path.expanduser(str(path))))
def resolve_model_roots(
explicit_roots: Optional[Iterable[Union[str, Path]]] = None,
*,
settings: Optional[AudioCppSettings] = None,
folder_paths_module=None,
) -> list[Path]:
"""Return search roots in priority order, with managed storage last."""
module = folder_paths_module if folder_paths_module is not None else _import_folder_paths()
current = settings or load_settings(folder_paths_module=module)
managed = default_managed_model_root(settings=current, folder_paths_module=module)
candidates: list[Path] = []
candidates.extend(Path(path).expanduser() for path in (explicit_roots or ()) if path)
candidates.extend(Path(path).expanduser() for path in current.model_roots if path)
candidates.extend(_registered_paths(module, "audio_cpp"))
candidates.extend(path / "audio.cpp" / "models" for path in _tts_paths(module))
managed_key = _canonical(managed)
seen: set[str] = set()
result: list[Path] = []
for candidate in candidates:
key = _canonical(candidate)
if key == managed_key or key in seen:
continue
seen.add(key)
result.append(candidate)
result.append(managed)
return result
def package_install_path(package: PackageRecord, models_root: Union[str, Path]) -> Path:
return Path(models_root) / Path(*package.target_directory.replace("\\", "/").split("/"))
def package_is_complete(package: PackageRecord, package_directory: Union[str, Path]) -> bool:
directory = Path(package_directory)
return directory.is_dir() and all(
(directory / path).is_file() and (directory / path).stat().st_size > 0
for path in package.local_files
)
def find_installed_package(
package: Union[str, PackageRecord],
roots: Optional[Sequence[Union[str, Path]]] = None,
*,
catalog: Optional[AudioCppCatalog] = None,
settings: Optional[AudioCppSettings] = None,
folder_paths_module=None,
) -> Optional[Path]:
current_catalog = catalog or load_catalog()
record = current_catalog.package(package) if isinstance(package, str) else package
search_roots = (
[Path(root) for root in roots]
if roots is not None
else resolve_model_roots(settings=settings, folder_paths_module=folder_paths_module)
)
for root in search_roots:
candidate = package_install_path(record, root)
if package_is_complete(record, candidate):
return candidate
return None
@dataclass(frozen=True)
class ResolvedModel:
package: PackageRecord
path: Path
root: Path
def resolve_model(
package: Union[str, PackageRecord],
roots: Optional[Sequence[Union[str, Path]]] = None,
*,
catalog: Optional[AudioCppCatalog] = None,
settings: Optional[AudioCppSettings] = None,
folder_paths_module=None,
) -> Optional[ResolvedModel]:
"""Resolve one catalog package and retain the root that won precedence."""
current_catalog = catalog or load_catalog()
record = current_catalog.package(package) if isinstance(package, str) else package
search_roots = (
[Path(root) for root in roots]
if roots is not None
else resolve_model_roots(settings=settings, folder_paths_module=folder_paths_module)
)
path = find_installed_package(record, search_roots, catalog=current_catalog)
if path is None:
return None
for root in search_roots:
if _canonical(package_install_path(record, root)) == _canonical(path):
return ResolvedModel(package=record, path=path, root=root)
return None
def discover_installed_packages(
roots: Optional[Sequence[Union[str, Path]]] = None,
*,
catalog: Optional[AudioCppCatalog] = None,
settings: Optional[AudioCppSettings] = None,
folder_paths_module=None,
) -> dict[str, ResolvedModel]:
current_catalog = catalog or load_catalog()
found: dict[str, ResolvedModel] = {}
for package in current_catalog.packages.values():
resolved = resolve_model(
package,
roots,
catalog=current_catalog,
settings=settings,
folder_paths_module=folder_paths_module,
)
if resolved is not None:
found[package.id] = resolved
return found
+306
View File
@@ -0,0 +1,306 @@
"""Direct, cache-free downloader for pinned audio.cpp model packages."""
from __future__ import annotations
import os
import shutil
import tempfile
import urllib.error
import urllib.request
import uuid
from dataclasses import dataclass
from pathlib import Path
from typing import Callable, Mapping, Optional, Union
from urllib.parse import quote
from .catalog import AudioCppCatalog, PackageRecord, load_catalog
from .discovery import package_install_path, package_is_complete
class AudioCppDownloadError(RuntimeError):
"""Raised when a model package cannot be downloaded safely."""
class IncompleteExistingPackageError(AudioCppDownloadError):
"""Raised when a package target exists but is not complete."""
ProgressCallback = Callable[[str, int, Optional[int]], None]
@dataclass(frozen=True)
class DownloadResult:
package: PackageRecord
path: Path
downloaded_files: tuple[Path, ...]
bytes_downloaded: int
already_present: bool = False
def resolve_hf_token(explicit_token: Optional[str] = None) -> Optional[str]:
if explicit_token:
return explicit_token.strip() or None
for name in ("HF_TOKEN", "HUGGING_FACE_HUB_TOKEN"):
value = os.environ.get(name, "").strip()
if value:
return value
try:
from huggingface_hub import get_token
value = get_token()
if value:
return value.strip()
except (ImportError, OSError, RuntimeError):
pass
token_path = Path.home() / ".cache" / "huggingface" / "token"
try:
value = token_path.read_text(encoding="utf-8").strip()
return value or None
except OSError:
return None
def huggingface_resolve_url(repo: str, revision: str, remote_path: str) -> str:
return (
"https://huggingface.co/"
f"{quote(repo, safe='/')}/resolve/{quote(revision, safe='')}/"
f"{quote(remote_path.replace(chr(92), '/'), safe='/')}"
)
def package_download_size(
package: Union[str, PackageRecord],
*,
catalog: Optional[AudioCppCatalog] = None,
token: Optional[str] = None,
timeout: int = 30,
opener=None,
) -> Optional[int]:
"""Return the declared HTTP size of every package file when available."""
current_catalog = catalog or load_catalog()
record = current_catalog.package(package) if isinstance(package, str) else package
headers = {"User-Agent": "TTS-Audio-Suite/audio.cpp-model-installer"}
hf_token = resolve_hf_token(token)
if hf_token:
headers["Authorization"] = f"Bearer {hf_token}"
open_request = opener or urllib.request.urlopen
total = 0
try:
for remote_path in record.files:
request = urllib.request.Request(
huggingface_resolve_url(record.repo, record.revision, remote_path),
headers=headers,
method="HEAD",
)
response = open_request(request, timeout=timeout)
try:
length = response.headers.get("Content-Length") if response.headers else None
if not length:
return None
total += int(length)
finally:
response.close()
except (OSError, TypeError, ValueError, urllib.error.URLError):
return None
return total or None
def download_url_to_path(
url: str,
destination: Union[str, Path],
*,
headers: Optional[Mapping[str, str]] = None,
timeout: int = 300,
opener=None,
progress: Optional[ProgressCallback] = None,
progress_label: str = "download",
) -> int:
"""Stream a URL to a new file and validate HTTP Content-Length."""
destination_path = Path(destination)
request = urllib.request.Request(url, headers=dict(headers or {}))
open_request = opener or urllib.request.urlopen
response = None
downloaded = 0
try:
response = open_request(request, timeout=timeout)
status = getattr(response, "status", None) or getattr(response, "code", None)
if status is not None and int(status) >= 400:
raise AudioCppDownloadError(f"HTTP {status} while downloading {url}")
content_length_value = response.headers.get("Content-Length") if response.headers else None
expected = int(content_length_value) if content_length_value else None
destination_path.parent.mkdir(parents=True, exist_ok=True)
with destination_path.open("xb") as handle:
while True:
chunk = response.read(1024 * 1024)
if not chunk:
break
handle.write(chunk)
downloaded += len(chunk)
if progress is not None:
progress(progress_label, downloaded, expected)
if expected is not None and downloaded != expected:
raise AudioCppDownloadError(
f"Incomplete download for {progress_label}: expected {expected} bytes, got {downloaded}"
)
return downloaded
except urllib.error.HTTPError as exc:
if exc.code in (401, 403):
raise AudioCppDownloadError(
f"Hugging Face denied {progress_label} (HTTP {exc.code}). "
"Accept any model license and configure HF_TOKEN."
) from exc
raise AudioCppDownloadError(f"HTTP {exc.code} while downloading {progress_label}") from exc
except urllib.error.URLError as exc:
raise AudioCppDownloadError(f"Network error downloading {progress_label}: {exc.reason}") from exc
except OSError as exc:
raise AudioCppDownloadError(f"Cannot write {destination_path}: {exc}") from exc
finally:
if response is not None:
try:
response.close()
except Exception:
pass
def _remove_path(path: Path) -> None:
if path.is_symlink() or path.is_file():
path.unlink(missing_ok=True)
elif path.is_dir():
shutil.rmtree(path)
def _link_or_copy(source: str, destination: str) -> str:
"""Hard-link existing model files into staging, copying only if necessary."""
try:
os.link(source, destination)
return destination
except OSError:
return shutil.copy2(source, destination)
def _merge_existing_target(target: Path, staging: Path) -> None:
"""Preserve other precision packages that share this target directory."""
if target.is_symlink() or not target.is_dir():
raise IncompleteExistingPackageError(
f"Model target cannot be safely merged because it is not a regular directory: {target}"
)
shutil.copytree(
target,
staging,
dirs_exist_ok=True,
copy_function=_link_or_copy,
symlinks=False,
)
def _publish_directory(staging: Path, target: Path, replace_existing: bool) -> None:
# PocketTTS targets are nested (for example PocketTTS-GGUF/english).
# The parent must exist before Windows can rename the staged directory.
target.parent.mkdir(parents=True, exist_ok=True)
if not target.exists() and not target.is_symlink():
staging.rename(target)
return
if not replace_existing:
raise IncompleteExistingPackageError(
f"Model target already exists but is incomplete: {target}. "
"Choose overwrite explicitly or repair it manually."
)
backup = target.with_name(f".{target.name}.{uuid.uuid4().hex}.backup")
target.rename(backup)
try:
staging.rename(target)
except BaseException:
if not target.exists() and backup.exists():
backup.rename(target)
raise
_remove_path(backup)
def install_package(
package: Union[str, PackageRecord],
models_root: Union[str, Path],
*,
catalog: Optional[AudioCppCatalog] = None,
overwrite: bool = False,
token: Optional[str] = None,
timeout: int = 300,
opener=None,
progress: Optional[ProgressCallback] = None,
) -> DownloadResult:
"""Download every required package file, then atomically publish it."""
current_catalog = catalog or load_catalog()
record = current_catalog.package(package) if isinstance(package, str) else package
if record.download.get("kind") != "huggingface_snapshot":
raise AudioCppDownloadError(
f"Unsupported download kind for {record.id}: {record.download.get('kind')!r}"
)
if not record.repo:
raise AudioCppDownloadError(f"Package {record.id!r} has no Hugging Face repository")
root = Path(models_root).expanduser()
target = package_install_path(record, root)
if package_is_complete(record, target) and not overwrite:
return DownloadResult(record, target, (), 0, already_present=True)
target_exists = target.exists() or target.is_symlink()
if target_exists:
if target.is_symlink() or not target.is_dir():
raise IncompleteExistingPackageError(
f"Model target cannot be safely merged because it is not a regular directory: {target}"
)
requested_files_exist = any(
(target / path).exists() or (target / path).is_symlink()
for path in record.local_files
)
if requested_files_exist and not overwrite:
raise IncompleteExistingPackageError(
f"Model target contains an incomplete package {record.id!r}: {target}. "
"Choose overwrite explicitly or repair it manually."
)
target.parent.mkdir(parents=True, exist_ok=True)
staging = Path(tempfile.mkdtemp(prefix=f".{target.name}.", suffix=".staging", dir=target.parent))
downloaded_files: list[Path] = []
downloaded_bytes = 0
hf_token = resolve_hf_token(token)
headers = {"User-Agent": "TTS-Audio-Suite/audio.cpp-model-installer"}
if hf_token:
headers["Authorization"] = f"Bearer {hf_token}"
try:
if target_exists:
_merge_existing_target(target, staging)
for remote_path, local_path in zip(record.files, record.local_files):
destination = staging / local_path
# The merge may have hard-linked an older copy of the requested
# precision. Remove that link before creating the replacement.
destination.unlink(missing_ok=True)
url = huggingface_resolve_url(record.repo, record.revision, remote_path)
downloaded_bytes += download_url_to_path(
url,
destination,
headers=headers,
timeout=timeout,
opener=opener,
progress=progress,
progress_label=remote_path,
)
downloaded_files.append(local_path)
if not package_is_complete(record, staging):
missing = [str(path) for path in record.local_files if not (staging / path).is_file()]
raise AudioCppDownloadError(
f"Package {record.id!r} staging validation failed; missing: {missing}"
)
_publish_directory(staging, target, replace_existing=target_exists)
return DownloadResult(
package=record,
path=target,
downloaded_files=tuple(downloaded_files),
bytes_downloaded=downloaded_bytes,
)
except BaseException:
_remove_path(staging)
raise
@@ -0,0 +1,232 @@
schema_version: 1
release: release-0.5.1
# Verified from Hugging Face file metadata on 2026-08-13. These are offline UI
# estimates; live Content-Length values remain authoritative during downloads.
package_sizes:
citrinet_asr_q8_0: 40574432
chatterbox_f16: 3744360386
chatterbox_q8_0: 2088393668
chatterbox_safetensors: 3191859618
confucius4_tts_orig: 8192757760
dramabox_q8_0: 18942803808
fish_audio_s2_pro_bf16: 10229278080
fish_audio_s2_pro_q8_0: 6317911232
fun_asr_nano_2512_f16: 1675708832
fun_asr_nano_2512_q8_0: 1045334432
fun_asr_nano_2512_safetensors: 1671205126
glm_tts_q8_0: 5143813728
higgs_audio_tts_4b_bf16: 8501587648
higgs_audio_tts_4b_q8_0: 5095354048
higgs_audio_stt_f16: 5367453248
higgs_audio_stt_q8_0: 3158310848
hviske_asr_q8_0: 2436359808
hviske_asr_safetensors: 4132293306
index_tts2_f16: 4646898304
index_tts2_orig: 8084552000
index_tts2_q8_0: 3633888608
index_tts2_safetensors: 3484956803
inflect_micro_v2_orig: 72082176
irodori_tts_500m_v3_f16: 1254813120
irodori_tts_500m_v3_q8_0: 1093739584
irodori_tts_600m_v3_voicedesign_f16: 1463787680
irodori_tts_600m_v3_voicedesign_q8_0: 1272140064
irodori_tts_v4_small_f16: 1762148352
irodori_tts_v4_small_q8_0: 1368991360
kroko_asr_community_q8_0: 167756928
miotts_1_7b_bf16: 3518532512
miotts_1_7b_orig: 3518532512
miotts_1_7b_q8_0: 2197326752
moss_tts_local_v1_5_bf16: 13367797280
moss_tts_local_v1_5_q8_0: 7512220768
moss_tts_nano_100m_bf16: 332423040
moss_tts_nano_100m_q8_0: 193337984
nemotron_asr_f16: 1277710880
nemotron_asr_q8_0: 930620256
nemotron_asr_safetensors: 2552818890
omnivoice_bf16: 1639548640
omnivoice_f16: 1639548768
omnivoice_q8_0: 1350288416
omnivoice_safetensors: 2461770336
outetts_1_0_1b_q8_0: 3029895456
parakeet_tdt_f16: 1255384320
parakeet_tdt_q8_0: 915733744
pocket_tts_english_bf16: 219096064
pocket_tts_english_q8_0: 127856704
pocket_tts_english_safetensors: 385107895
pocket_tts_german_bf16: 219096544
pocket_tts_german_q8_0: 127857184
pocket_tts_italian_bf16: 219096800
pocket_tts_italian_q8_0: 127857440
pocket_tts_portuguese_bf16: 219097728
pocket_tts_portuguese_q8_0: 127858368
pocket_tts_spanish_bf16: 219097600
pocket_tts_spanish_q8_0: 127858240
qwen3_tts_0_6b_base_safetensors: 2511651783
qwen3_tts_1_7b_base_bf16: 4203158464
qwen3_tts_1_7b_base_orig: 4544273280
qwen3_tts_1_7b_base_q8_0: 2695175104
qwen3_tts_1_7b_base_safetensors: 4539721255
qwen3_tts_1_7b_customvoice_bf16: 4179144352
qwen3_tts_1_7b_customvoice_q8_0: 2817044064
qwen3_tts_1_7b_voicedesign_bf16: 4179089248
qwen3_tts_1_7b_voicedesign_q8_0: 2816988960
qwen3_asr_0_6b_f16: 1880642016
qwen3_asr_0_6b_q8_0: 1151272416
qwen3_asr_0_6b_safetensors: 1876110856
qwen3_asr_1_7b_f16: 4087653248
qwen3_asr_1_7b_q8_0: 2473010048
qwen3_asr_1_7b_safetensors: 4087626782
seed_vc_mlx_f16: 3629186560
seed_vc_mlx_orig: 7003311936
seed_vc_mlx_q8_0: 3120809248
seed_vc_mlx_safetensors: 712430273
supertonic_3_f16: 312784196
supertonic_3_orig: 454072836
supertonic_3_q8_0: 454072836
supertonic_3_safetensors: 397279577
vevo2_f16: 5021112128
vevo2_orig: 7449615488
vevo2_q8_0: 3242124800
vibevoice_1_5b_bf16: 5420021858
vibevoice_1_5b_q8_0: 3224701538
vibevoice_asr_f16: 17361090304
vibevoice_asr_q8_0: 9858644224
vietneu_tts_v3_turbo_q8_0: 170482368
voxcpm2_bf16: 4772288288
voxcpm2_orig: 4960716192
voxcpm2_q8_0: 2955000480
voxcpm2_safetensors: 4583766759
voxtral_realtime_bf16: 8874402784
voxtral_realtime_q4_k: 3097662432
voxtral_realtime_q8_0: 5104567264
# Runtime dependencies omitted by upstream TTS package declarations. These are
# installed independently and passed to audio.cpp through absolute session paths.
package_dependencies:
miotts_1_7b_q8_0: &miotts_codec_dependency
- id: miocodec_q8_0
family: miocodec
display_name: "MioCodec 25Hz 44.1kHz v2 Q8_0 GGUF"
target_directory: "MioCodec-25Hz-44.1kHz-v2-GGUF"
format: gguf
precision: q8_0
files: ["MioCodec-25Hz-44.1kHz-v2-GGUF/miocodec-25hz-44khz-v2-q8_0.gguf"]
strip_prefix: "MioCodec-25Hz-44.1kHz-v2-GGUF"
repo: "audio-cpp/audio.cpp-gguf"
revision: main
estimated_download_bytes: 299066464
session_option: "miotts.codec_model_path"
miotts_1_7b_bf16: *miotts_codec_dependency
miotts_1_7b_orig: *miotts_codec_dependency
# Suite-owned UI and validation metadata. Download/package truth remains in
# model_specs/*.json, which are unmodified snapshots of audio.cpp.
defaults:
suite_support: supported
suite_tasks: [tts, srt, character_switching]
reference_audio: none
reference_transcript: none
built_in_voices: false
voice_design: false
inline_controls: false
native_multi_speaker: {supported: false, max_speakers: 1, suite_status: unavailable}
asr_features: {diarization: none, timing: none}
families:
citrinet_asr:
suite_tasks: [asr]
summary: "Compact English transcription using a Citrinet CTC model."
chatterbox:
reference_audio: required
suite_tasks: [tts, srt, character_switching, voice_conversion]
summary: "Voice cloning and reference-targeted voice conversion."
confucius4_tts: {reference_audio: required, summary: "Voice cloning from reference audio."}
dramabox: {reference_audio: optional, summary: "TTS and optional voice cloning."}
fish_audio: {reference_audio: optional, summary: "Reference-conditioned TTS."}
fun_asr_nano:
suite_tasks: [asr]
summary: "Multilingual speech recognition for Chinese, English, and Japanese."
glm_tts:
reference_audio: required
reference_transcript: required
summary: "Zero-shot cloning; reference audio and its transcript are required."
higgs_audio_tts:
reference_audio: optional
reference_transcript: optional
inline_controls: true
summary: "Expressive multilingual TTS, cloning, and inline style/sound controls."
higgs_audio_stt:
suite_tasks: [asr]
summary: "English transcription using the Higgs Audio v3 speech model."
hviske_asr:
suite_tasks: [asr]
summary: "Multilingual transcription across 16 languages."
index_tts2: {reference_audio: optional, summary: "TTS with optional voice cloning."}
inflect_v2: {summary: "Reference-free expressive TTS."}
irodori_tts:
reference_audio: optional
voice_design: true
summary: "TTS, optional cloning, and voice design."
kroko_asr:
suite_tasks: [asr]
asr_features: {timing: native_word}
summary: "Small multilingual community ASR model."
miotts: {reference_audio: optional, summary: "Reference-conditioned TTS."}
moss_tts_local: {reference_audio: optional, summary: "TTS with optional voice cloning."}
moss_tts_nano: {reference_audio: optional, summary: "Compact TTS with optional cloning."}
nemotron_asr:
suite_tasks: [asr]
asr_features: {timing: native_word}
summary: "Streaming-capable multilingual NVIDIA Nemotron transcription."
omnivoice:
reference_audio: optional
reference_transcript: optional
voice_design: true
summary: "Reference cloning or instruction-based voice design."
outetts:
reference_audio: optional
reference_transcript: optional
summary: "Multilingual TTS; a transcript improves reference cloning."
parakeet_tdt:
suite_tasks: [asr]
asr_features: {timing: native_word}
summary: "Fast multilingual Parakeet-TDT transcription."
pocket_tts: {reference_audio: optional, summary: "Small CPU-friendly TTS and cloning."}
qwen3_tts:
reference_audio: optional
voice_design: true
summary: "TTS with package-dependent cloning or voice design."
qwen3_asr:
suite_tasks: [asr]
asr_features: {timing: optional_forced_aligner}
summary: "Broad multilingual Qwen3 transcription; word timestamps require an optional forced aligner."
seed_vc:
suite_tasks: [voice_conversion]
reference_audio: required
summary: "Reference-targeted voice conversion through Seed-VC."
supertonic:
built_in_voices: true
summary: "Fast multilingual TTS using built-in voices."
vevo2:
reference_audio: optional
suite_tasks: [tts, srt, character_switching, voice_conversion]
summary: "TTS and reference-targeted voice conversion; S2S/SVC remain advanced upstream routes."
vibevoice:
reference_audio: required_per_speaker
native_multi_speaker: {supported: true, max_speakers: 4, suite_status: partial}
summary: "Long-form dialogue for up to four speakers. Native single-request mode is not wired yet."
vibevoice_asr:
suite_tasks: [asr, diarization]
asr_features: {diarization: native, timing: native_segment}
summary: "Multilingual transcription with native timestamped speaker turns."
vietneu_tts:
voice_design: true
summary: "Vietnamese TTS and voice design."
voxcpm2:
reference_audio: optional
voice_design: true
summary: "TTS with reference conditioning and voice design."
voxtral_realtime:
suite_tasks: [asr]
summary: "Multilingual realtime-oriented Voxtral transcription."
+287
View File
@@ -0,0 +1,287 @@
"""ComfyUI model-management proxy for suite-owned audio.cpp processes."""
from __future__ import annotations
import os
import sys
import threading
import weakref
from pathlib import Path
from typing import TYPE_CHECKING, Any, Dict, Mapping, Optional
if TYPE_CHECKING:
from .session import AudioCppSession
def _first(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
for key in keys:
if key in config and config[key] is not None:
return config[key]
return default
def _warn(message: str, exc: Optional[BaseException] = None) -> None:
text = f"WARNING: {message}"
if exc is not None:
text += f": {exc}"
encoding = getattr(sys.stderr, "encoding", None) or "ascii"
try:
text = text.encode(encoding, errors="replace").decode(encoding, errors="replace")
except LookupError:
text = text.encode("ascii", errors="replace").decode("ascii")
print(text, file=sys.stderr)
class AudioCppRuntimeProxy:
"""ComfyUI model-like resource representing one owned native server."""
def __init__(self, session: "AudioCppSession") -> None:
self._session_ref = weakref.ref(session)
self.model = self
self.processor = self
self.parent = None
self.currently_used = True
self.model_options: Dict[str, Any] = {}
self.model_keys = set()
self.offload_device = self._torch_device("cpu", 0)
backend = str(session.config.get("backend", "cuda")).lower()
device_index = self._device_index(
session.config.get("device_index", session.config.get("device", 0))
)
self.load_device = self._lifecycle_device(backend, device_index)
self.current_device = self.load_device
self._estimated_memory_size = self._memory_estimate(session.config)
self._loaded_model = None
self._model_management = None
self._registration_lock = threading.RLock()
@staticmethod
def _device_index(value: Any) -> int:
text = str(value or 0).lower()
if ":" in text:
text = text.rsplit(":", 1)[-1]
try:
return max(0, int(text))
except ValueError:
return 0
@staticmethod
def _torch_device(kind: str, index: int):
try:
import torch
return torch.device(kind, index) if kind == "cuda" else torch.device(kind)
except Exception:
return f"{kind}:{index}" if kind == "cuda" else kind
@classmethod
def _lifecycle_device(cls, backend: str, index: int):
if backend == "cpu":
return cls._torch_device("cpu", 0)
try:
import comfy.model_management as model_management
device = model_management.get_torch_device()
if device is not None:
return device
except (ImportError, AttributeError, RuntimeError):
pass
if backend == "metal":
return cls._torch_device("mps", 0)
return cls._torch_device("cuda", index)
@staticmethod
def _memory_estimate(config: Mapping[str, Any]) -> int:
explicit = _first(config, "estimated_memory_bytes", "model_memory_bytes")
if explicit is not None:
return max(1, int(explicit))
gigabytes = _first(config, "estimated_vram_gb", "model_memory_gb")
if gigabytes is not None:
return max(1, int(float(gigabytes) * 1024**3))
raw_path = _first(config, "model_path", "package_path", "gguf_path")
if raw_path:
try:
model_path = Path(
os.path.expandvars(os.path.expanduser(str(raw_path)))
).resolve()
if model_path.is_file():
return max(1, model_path.stat().st_size)
if model_path.is_dir():
total = 0
for candidate in model_path.rglob("*"):
try:
if candidate.is_file():
total += candidate.stat().st_size
except OSError:
continue
return max(1, total)
except OSError:
pass
# A zero-size entry is ignored by parts of ComfyUI's unload ordering.
return 1
def _session(self) -> Optional["AudioCppSession"]:
return self._session_ref()
def register(self) -> bool:
"""Register once with ComfyUI after an owned process becomes live."""
with self._registration_lock:
if self._loaded_model is not None:
current = getattr(self._model_management, "current_loaded_models", None)
if isinstance(current, list) and self._loaded_model in current:
return True
self._loaded_model = None
self._model_management = None
try:
import comfy.model_management as model_management
except ImportError:
return False
try:
loaded_model_type = getattr(model_management, "LoadedModel", None)
current_models = getattr(model_management, "current_loaded_models", None)
if not callable(loaded_model_type) or not isinstance(current_models, list):
return False
loaded_model = loaded_model_type(self)
loaded_model.real_model = weakref.ref(self)
cleanup = getattr(model_management, "cleanup_models", None)
loaded_model.model_finalizer = weakref.finalize(
self, cleanup if callable(cleanup) else lambda: None
)
loaded_model._tts_wrapper_ref = self
current_models.insert(0, loaded_model)
self._loaded_model = loaded_model
self._model_management = model_management
return True
except Exception as exc:
_warn("Failed to register audio.cpp runtime with ComfyUI", exc)
return False
def unregister(self) -> None:
with self._registration_lock:
loaded_model = self._loaded_model
model_management = self._model_management
self._loaded_model = None
self._model_management = None
if loaded_model is None or model_management is None:
return
current_models = getattr(model_management, "current_loaded_models", None)
if not isinstance(current_models, list):
return
try:
for candidate in list(current_models):
if candidate is loaded_model or getattr(candidate, "_tts_wrapper_ref", None) is self:
current_models.remove(candidate)
except Exception as exc:
_warn("Failed to unregister audio.cpp runtime from ComfyUI", exc)
def to(self, device):
self.current_device = device
return self
def eval(self):
return self
def model_size(self) -> int:
return self._estimated_memory_size
def loaded_size(self) -> int:
session = self._session()
return self._estimated_memory_size if session is not None and session.running else 0
def model_memory(self) -> int:
return self.model_size()
def get_ram_usage(self) -> int:
return self._estimated_memory_size
def model_offloaded_memory(self) -> int:
return max(0, self.model_size() - self.loaded_size())
def model_mmap_residency(self, free: bool = False) -> tuple[int, int]:
return 0, self._estimated_memory_size
def pinned_memory_size(self) -> int:
return 0
def lowvram_patch_counter(self) -> int:
return 0
def model_dtype(self):
try:
import torch
return torch.float32
except Exception:
return None
def current_loaded_device(self):
return self.current_device
def model_patches_models(self):
return ()
def model_patches_to(self, target) -> None:
try:
import torch
if isinstance(target, torch.device):
self.current_device = target
except Exception:
pass
def is_dynamic(self) -> bool:
return False
def is_clone(self, other) -> bool:
return other is self
def clone_has_same_weights(self, other) -> bool:
return other is self
def partially_load(self, device, extra_memory, force_patch_weights=False) -> int:
self.current_device = device
return 0
def partially_unload(self, device, memory_to_free) -> int:
# audio.cpp 0.5.1 cannot release part of a session. Claiming memory here
# would make ComfyUI believe VRAM was freed while the server still owns it.
return 0
def partially_unload_ram(self, ram_to_unload) -> int:
return 0
def patch_model(
self,
device_to=None,
lowvram_model_memory=0,
load_weights=True,
force_patch_weights=False,
):
if device_to is not None:
self.current_device = device_to
return self.model
def unpatch_model(self, device_to=None, unpatch_weights=True):
if device_to is not None:
self.current_device = device_to
session = self._session()
if session is not None:
# LoadedModel removes its list entry after this callback returns.
session._stop_owned_runtime(unregister=False)
return self.model
def model_unload(self, memory_to_free=None, unpatch_weights=True) -> bool:
self.unpatch_model(self.offload_device, unpatch_weights=unpatch_weights)
return True
def detach(self, unpatch_weights=True):
return self.unpatch_model(self.offload_device, unpatch_weights=unpatch_weights)
def cleanup(self) -> None:
session = self._session()
if session is not None:
session._stop_owned_runtime(unregister=True)
__all__ = ["AudioCppRuntimeProxy"]
+168
View File
@@ -0,0 +1,168 @@
{
"family": "chatterbox",
"display_name": "Chatterbox",
"description": "Open-source Chatterbox family for expressive TTS and voice conversion, with emotion exaggeration control, fast generation, zero-shot voice cloning, and an integrated multilingual TTS path.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone",
"vc"
],
"modes": [
"offline"
],
"languages": [
"ar",
"da",
"de",
"el",
"en",
"es",
"fi",
"fr",
"hi",
"it",
"ko",
"ms",
"nl",
"no",
"pl",
"pt",
"sv",
"sw",
"tr"
],
"capabilities": {
"clone": [
"speaker_reference"
],
"vc": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "chatterbox_q8_0",
"tags": [
"TTS",
"Clone",
"VC",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "chatterbox_q8_0",
"display_name": "Chatterbox Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Chatterbox-GGUF",
"files": [
"Chatterbox-GGUF/chatterbox-q8_0.gguf"
],
"strip_prefix": "Chatterbox-GGUF"
},
{
"id": "chatterbox_f16",
"display_name": "Chatterbox F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Chatterbox-GGUF",
"files": [
"Chatterbox-GGUF/chatterbox-f16.gguf"
],
"strip_prefix": "Chatterbox-GGUF"
},
{
"id": "chatterbox_safetensors",
"display_name": "Chatterbox Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "chatterbox",
"files": [
"ve.safetensors",
"t3_cfg.safetensors",
"s3gen.safetensors",
"tokenizer.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "ResembleAI/chatterbox"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"english_tokenizer": "model:tokenizer.json",
"multilingual_tokenizer": "model:grapheme_mtl_merged_expanded_v1.json",
"cangjie_mapping": "model:Cangjie5_TC.json",
"builtin_conditionals": "model:conds.pt"
},
"tensors": {
"voice_encoder_weights": {
"source": "weights:",
"prefix": "voice_encoder"
},
"s3gen_weights": {
"source": "weights:",
"prefix": "s3gen"
},
"t3_english_weights": {
"source": "weights:",
"prefix": "t3_english"
},
"t3_multilingual_v2_weights": {
"source": "weights:",
"prefix": "t3_multilingual_v2"
},
"t3_multilingual_v3_weights": {
"source": "weights:",
"prefix": "t3_multilingual_v3"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"english_tokenizer": "model:tokenizer.json",
"multilingual_tokenizer": "model:grapheme_mtl_merged_expanded_v1.json",
"cangjie_mapping": "model:Cangjie5_TC.json",
"builtin_conditionals": "model:conds.pt"
},
"tensors": {
"voice_encoder_weights": "model:ve.safetensors",
"s3gen_weights": "model:s3gen.safetensors",
"t3_english_weights": "model:t3_cfg.safetensors",
"t3_multilingual_v2_weights": "model:t3_mtl23ls_v2.safetensors",
"t3_multilingual_v3_weights": "model:t3_mtl23ls_v3.safetensors"
}
}
]
}
@@ -0,0 +1,84 @@
{
"family": "citrinet_asr",
"display_name": "Citrinet ASR",
"description": "NVIDIA CitriNet-family end-to-end English ASR model using a convolutional CTC architecture optimized for transcribing speech segments to text.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline"
],
"languages": [
"en"
],
"capabilities": {},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "citrinet_asr_q8_0",
"tags": [
"ASR",
"GGUF"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "citrinet_asr_q8_0",
"display_name": "Citrinet ASR Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Citrinet-ASR-GGUF",
"files": [
"Citrinet-ASR-GGUF/citrinet-asr-q8_0.gguf"
],
"strip_prefix": "Citrinet-ASR-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:citrinet_256_config.json",
"tokenizer": "model:citrinet_256_tokenizer.model"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:citrinet_256_config.json",
"tokenizer": "model:citrinet_256_tokenizer.model"
},
"tensors": {
"weights": "model:citrinet_256.safetensors"
}
}
]
}
@@ -0,0 +1,330 @@
{
"schema_version": 1,
"family": "confucius4_tts",
"display_name": "Confucius4-TTS",
"description": "Confucius4-TTS is a multilingual voice-cloning TTS model packaged for audio.cpp with offline and streaming generation. It uses reference speech, language-aware text normalization, T2S semantic generation, S2A flow matching, style encoding, semantic audio features, and BigVGAN vocoding.",
"category": "tts",
"status": "experimental",
"tasks": [
"clone"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"zh",
"en",
"ja",
"ko",
"de",
"fr",
"es",
"id",
"it",
"th",
"pt",
"ru",
"ms",
"vi"
],
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"capabilities": {
"clone": [
"speaker_reference",
"long_form"
]
},
"options": {
"request": [
{
"name": "language",
"type": "string",
"description": "Target synthesis language code used by the text frontend; default zh when no request transcript language or style language is provided.",
"required": false,
"default": "zh"
},
{
"name": "temperature",
"type": "float",
"description": "T2S sampling temperature; must be positive, default 0.8.",
"required": false,
"min": 0.0,
"default": 0.8
},
{
"name": "top_p",
"type": "float",
"description": "T2S nucleus sampling probability; must be in (0, 1], default 0.8.",
"required": false,
"min": 0.0,
"max": 1.0,
"default": 0.8
},
{
"name": "top_k",
"type": "int",
"description": "T2S top-k sampling limit; must be positive, default 30.",
"required": false,
"min": 1,
"default": 30
},
{
"name": "num_beams",
"type": "int",
"description": "T2S beam count; default 3. Set 1 for single-beam sampling.",
"required": false,
"min": 1,
"default": 3
},
{
"name": "repetition_penalty",
"type": "float",
"description": "T2S repetition penalty; must be positive, default 10.0.",
"required": false,
"min": 0.0,
"default": 10.0
},
{
"name": "max_tokens",
"type": "int",
"description": "Maximum T2S semantic sequence length including prompt tokens; default 1520.",
"required": false,
"min": 1,
"default": 1520
},
{
"name": "num_inference_steps",
"type": "int",
"description": "S2A flow-matching step count; default 25.",
"required": false,
"min": 1,
"default": 25
},
{
"name": "guidance_scale",
"type": "float",
"description": "S2A classifier-free guidance scale; default 0.7.",
"required": false,
"min": 0.0,
"default": 0.7
},
{
"name": "text_chunk_size",
"type": "int",
"description": "Maximum text tokens per generated segment; default 80.",
"required": false,
"min": 1,
"default": 80
},
{
"name": "text_chunk_mode",
"type": "enum",
"description": "Framework text chunking mode; default uses the standard word-budget chunker.",
"values": [
"default",
"tag_aware",
"japanese",
"endline"
],
"required": false,
"default": "default"
},
{
"name": "cross_fade_duration_sec",
"type": "float",
"description": "Cross-fade duration between generated segments in seconds; default 0.3.",
"required": false,
"min": 0.0,
"default": 0.3
},
{
"name": "edge_fade_duration_sec",
"type": "float",
"description": "Fade duration applied at segment edges in seconds; default 0.1.",
"required": false,
"min": 0.0,
"default": 0.1
},
{
"name": "edge_pad_duration_sec",
"type": "float",
"description": "Silence padding applied at segment edges in seconds; default 0.1.",
"required": false,
"min": 0.0,
"default": 0.1
},
{
"name": "seed",
"type": "int",
"description": "Seed for T2S sampling and S2A noise initialization; default 1234.",
"required": false,
"min": 0,
"default": 1234
}
],
"session": [
{
"name": "graph_arena_mb",
"type": "int",
"description": "Reusable ggml graph arena size in MiB for Confucius stages; default 512.",
"required": false,
"min": 1,
"default": 512
},
{
"name": "weight_context_mb",
"type": "int",
"description": "Weight loading context size in MiB; default 1024.",
"required": false,
"min": 1,
"default": 1024
},
{
"name": "weight_type",
"type": "enum",
"description": "Matmul weight storage type; default native.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "conv_weight_type",
"type": "enum",
"description": "Convolution weight storage type; default native.",
"preset": "weight_type_conv",
"required": false,
"default": "native"
},
{
"name": "reference_cache_slots",
"type": "int",
"description": "Prepared reference-audio cache slots; default 1, set 0 to disable caching.",
"required": false,
"min": 0,
"default": 1
},
{
"name": "mem_saver",
"type": "bool",
"description": "Release staged graphs after request phases; default false.",
"required": false,
"default": false
}
],
"load": []
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "confucius4_tts_orig",
"display_name": "Confucius4-TTS Original-Dtype GGUF",
"default": true,
"format": "gguf",
"precision": "orig",
"target_directory": "Confucius4-TTS-GGUF",
"files": [
"Confucius4-TTS-GGUF/confucius4-tts-orig.gguf"
],
"strip_prefix": "Confucius4-TTS-GGUF"
}
],
"dependencies": [],
"ui": {
"recommended_package": "confucius4_tts_orig",
"tags": [
"TTS",
"Clone",
"GGUF",
"Stream"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:inference_config.yaml",
"tokenizer_model": "model:tokenizer.model",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"special_tokens_map": "model:special_tokens_map.json",
"w2v_preprocessor_config": "model:w2v_preprocessor_config.json",
"bigvgan_config": "model:bigvgan_config.json"
},
"tensors": {
"t2s": {
"source": "weights:",
"prefix": "t2s"
},
"s2a": {
"source": "weights:",
"prefix": "s2a"
},
"semantic_encoder": {
"source": "weights:",
"prefix": "semantic_encoder"
},
"semantic_encoder_shaw": {
"source": "weights:",
"prefix": "semantic_encoder_shaw"
},
"semantic_stats": {
"source": "weights:",
"prefix": "semantic_stats"
},
"style_encoder": {
"source": "weights:",
"prefix": "style_encoder"
},
"vocoder": {
"source": "weights:",
"prefix": "vocoder"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:inference_config.yaml",
"tokenizer_model": "model:tokenizer.model",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"special_tokens_map": "model:special_tokens_map.json",
"w2v_preprocessor_config": "model:w2v_preprocessor_config.json",
"bigvgan_config": "model:bigvgan_config.json"
},
"tensors": {
"t2s": "model:t2s.safetensors",
"s2a": "model:s2a.safetensors",
"semantic_encoder": "model:semantic_encoder.safetensors",
"semantic_encoder_shaw": "model:semantic_encoder_shaw.safetensors",
"semantic_stats": "model:semantic_stats.safetensors",
"style_encoder": "model:style_encoder.safetensors",
"vocoder": "model:vocoder.safetensors"
}
}
]
}
+229
View File
@@ -0,0 +1,229 @@
{
"schema_version": 1,
"family": "dramabox",
"display_name": "DramaBox",
"description": "DramaBox is an English expressive TTS and voice-cloning model packaged for audio.cpp as a standalone GGUF bundle. It combines Gemma text conditioning, diffusion sampling, reference-audio conditioning, long-form chunking, and 48 kHz stereo output.",
"category": "tts",
"status": "experimental",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"en"
],
"runtime": {
"tags": [
"gguf"
]
},
"capabilities": {
"tts": [
"speaker_reference",
"style_control",
"long_form"
],
"clone": [
"speaker_reference",
"style_control",
"long_form"
]
},
"options": {
"request": [
{
"name": "target_voice",
"type": "audio_path",
"description": "Reference voice WAV path for voice cloning. If omitted, DramaBox generates from text without reference-audio conditioning.",
"required": false
},
{
"name": "negative_prompt",
"type": "string",
"description": "Negative text conditioning used when guidance_scale enables classifier-free guidance; omitted uses the built-in quality prompt.",
"required": false
},
{
"name": "duration_sec",
"type": "float",
"description": "Explicit target duration in seconds; default 0 uses the prompt-duration estimator.",
"required": false,
"min": 0.0,
"default": 0.0
},
{
"name": "num_inference_steps",
"type": "int",
"description": "Diffusion sampling step count; default comes from config.json, 30 in the current package.",
"required": false,
"min": 1,
"default": 30
},
{
"name": "guidance_scale",
"type": "float",
"description": "Classifier-free guidance scale; default comes from config.json, 2.5 in the current package. Values greater than 1 enable CFG.",
"required": false,
"min": 0.0,
"default": 2.5
},
{
"name": "spatio_temporal_guidance_scale",
"type": "float",
"description": "Spatio-temporal guidance scale; default comes from config.json, 1.5 in the current package. Values greater than 0 enable STG.",
"required": false,
"min": 0.0,
"default": 1.5
},
{
"name": "duration_scale",
"type": "float",
"description": "Multiplier applied to the estimated prompt duration when duration_sec is 0; default comes from config.json, 1.1 in the current package.",
"required": false,
"min": 0.0,
"default": 1.1
},
{
"name": "reference_duration_sec",
"type": "float",
"description": "Reference voice crop/repeat duration in seconds; default comes from config.json, 10.0 in the current package.",
"required": false,
"min": 0.0,
"default": 10.0
},
{
"name": "guidance_rescale",
"type": "string",
"description": "Guidance rescale value. The default auto mode derives a rescale value from guidance_scale; a numeric string requests an explicit value.",
"required": false,
"default": "auto"
},
{
"name": "audio_chunk_threshold_sec",
"type": "float",
"description": "Estimated duration threshold that switches a request to long-form chunking; default 45.0 seconds.",
"required": false,
"min": 0.0,
"default": 45.0
},
{
"name": "audio_chunk_duration_sec",
"type": "float",
"description": "Target estimated duration for each long-form chunk; default 37.0 seconds.",
"required": false,
"min": 0.0,
"default": 37.0
},
{
"name": "cross_fade_duration_sec",
"type": "float",
"description": "Equal-power cross-fade between long-form chunks in seconds; default 0.05.",
"required": false,
"min": 0.0,
"default": 0.05
},
{
"name": "seed",
"type": "int",
"description": "Torch-compatible CUDA noise seed for diffusion sampling; default 42.",
"required": false,
"min": 0,
"default": 42
}
],
"session": [
{
"name": "perf_mode",
"type": "enum",
"description": "Attention implementation mode. Default off keeps the exact reference-query attention path; flash_attention enables the optimized path.",
"values": [
"off",
"flash_attention"
],
"required": false,
"default": "off"
},
{
"name": "mem_saver",
"type": "bool",
"description": "Release staged runtime graphs and weights immediately after each request phase to reduce peak and resident VRAM; default false keeps components cached for later reuse.",
"required": false,
"default": false
}
],
"load": []
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "dramabox_q8_0",
"display_name": "DramaBox Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "DramaBox-GGUF",
"files": [
"DramaBox-GGUF/dramabox-q8_0.gguf"
],
"strip_prefix": "DramaBox-GGUF"
}
],
"dependencies": [],
"ui": {
"recommended_package": "dramabox_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"audio_components_config": "model:audio_components_config.json",
"gemma_config": "model:gemma-3-12b-it-bnb-4bit/config.json",
"gemma_tokenizer_model": "model:gemma-3-12b-it-bnb-4bit/tokenizer.model",
"gemma_tokenizer_json": "model:gemma-3-12b-it-bnb-4bit/tokenizer.json",
"gemma_tokenizer_config": "model:gemma-3-12b-it-bnb-4bit/tokenizer_config.json"
},
"tensors": {
"dit_weights": {
"source": "weights:",
"prefix": "dit"
},
"audio_weights": {
"source": "weights:",
"prefix": "audio"
},
"gemma_weights": {
"source": "weights:",
"prefix": "gemma"
},
"silence_latent": {
"source": "weights:",
"prefix": "silence"
}
}
}
]
}
+111
View File
@@ -0,0 +1,111 @@
{
"family": "fish_audio",
"display_name": "Fish Audio S2 Pro",
"description": "Fish Audio S2 Pro text-to-speech model for expressive speech across 80+ languages, with automatic language handling, inline prosody/emotion controls, and rapid voice cloning from short reference samples.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"80+ languages"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "fish_audio_s2_pro_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "fish_audio_s2_pro_q8_0",
"display_name": "Fish Audio S2 Pro Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Fish-Audio-S2-Pro-GGUF",
"files": [
"Fish-Audio-S2-Pro-GGUF/fish-audio-s2-pro-q8_0.gguf"
],
"strip_prefix": "Fish-Audio-S2-Pro-GGUF"
},
{
"id": "fish_audio_s2_pro_bf16",
"display_name": "Fish Audio S2 Pro BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "Fish-Audio-S2-Pro-GGUF",
"files": [
"Fish-Audio-S2-Pro-GGUF/fish-audio-s2-pro-bf16.gguf"
],
"strip_prefix": "Fish-Audio-S2-Pro-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"codec_weights": {
"source": "weights:",
"prefix": "codec_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"model_weights": "model:model_audio_cpp.safetensors.index.json",
"codec_weights": "model:codec.safetensors"
}
}
]
}
@@ -0,0 +1,201 @@
{
"schema_version": 1,
"family": "fun_asr_nano",
"display_name": "Fun-ASR-Nano",
"description": "Offline multilingual speech recognition with the FunAudioLLM Fun-ASR-Nano-2512 model.",
"category": "asr",
"status": "wip",
"tasks": [
"asr"
],
"modes": [
"offline"
],
"languages": [
"auto",
"zh",
"en",
"ja"
],
"capabilities": {},
"options": {
"request": [
{
"name": "language",
"type": "string",
"description": "Recognition language, or auto to let the model infer it.",
"required": false,
"default": "auto"
},
{
"name": "enable_itn",
"type": "bool",
"description": "Enable inverse text normalization in the transcription prompt.",
"required": false,
"default": true
},
{
"name": "max_tokens",
"type": "int",
"description": "Maximum number of generated transcript tokens.",
"required": false,
"min": 1,
"default": 512
},
{
"name": "audio_chunk_mode",
"type": "enum",
"description": "Audio chunking mode: auto, fixed, or none.",
"values": [
"auto",
"fixed",
"none"
],
"required": false,
"default": "auto"
},
{
"name": "audio_chunk_seconds",
"type": "float",
"description": "Fixed chunk duration in seconds.",
"required": false,
"min": 0.001,
"default": 30
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Shared model weight storage type.",
"preset": "weight_type_full",
"required": false,
"default": "native"
}
],
"load": []
},
"runtime": {
"tags": [
"gguf",
"server",
"cuda",
"metal",
"cpu"
]
},
"packages": [
{
"id": "fun_asr_nano_2512_q8_0",
"display_name": "Fun-ASR-Nano-2512 Q8_0 GGUF",
"description": "Standalone audio.cpp GGUF built from the pinned official checkpoint; governed by the FunASR Model Open Source License Agreement v1.1.",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Fun-ASR-Nano-2512-GGUF",
"files": [
"fun-asr-nano-2512-q8_0.gguf"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "FunAudioLLM/Fun-ASR-Nano-2512-GGUF",
"revision": "ce72677f84900f0dc57f498ace253bfb3c9155b6",
"gated": false
}
},
{
"id": "fun_asr_nano_2512_f16",
"display_name": "Fun-ASR-Nano-2512 F16 GGUF",
"description": "Standalone audio.cpp GGUF built from the pinned official checkpoint; governed by the FunASR Model Open Source License Agreement v1.1.",
"format": "gguf",
"precision": "f16",
"target_directory": "Fun-ASR-Nano-2512-GGUF",
"files": [
"fun-asr-nano-2512-f16.gguf"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "FunAudioLLM/Fun-ASR-Nano-2512-GGUF",
"revision": "ce72677f84900f0dc57f498ace253bfb3c9155b6",
"gated": false
}
},
{
"id": "fun_asr_nano_2512_safetensors",
"display_name": "Fun-ASR-Nano-2512 HF Safetensors",
"description": "Official checkpoint governed by the FunASR Model Open Source License Agreement v1.1.",
"format": "safetensors",
"precision": "native",
"target_directory": "Fun-ASR-Nano-2512-hf",
"files": [
"chat_template.jinja",
"config.json",
"generation_config.json",
"model.safetensors",
"processor_config.json",
"tokenizer.json",
"tokenizer_config.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "FunAudioLLM/Fun-ASR-Nano-2512-hf",
"revision": "854d88f94205cd17d2afdb24332130d86fbe654a",
"gated": false
}
}
],
"dependencies": [],
"ui": {
"recommended_package": "fun_asr_nano_2512_q8_0",
"tags": [
"ASR",
"GGUF"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
],
"summary": "Offline Fun-ASR-Nano transcription from official safetensors or audio.cpp GGUF."
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"processor_config": "model:processor_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"optional_files": {
"chat_template_jinja": "model:chat_template.jinja",
"tokenizer_config": "model:tokenizer_config.json"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"processor_config": "model:processor_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"optional_files": {
"chat_template_jinja": "model:chat_template.jinja",
"tokenizer_config": "model:tokenizer_config.json"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
+260
View File
@@ -0,0 +1,260 @@
{
"schema_version": 1,
"family": "glm_tts",
"display_name": "GLM-TTS",
"description": "Community Chinese-English zero-shot speech synthesis and voice cloning with native Llama, Whisper-VQ, Flow/DiT, CAMPPlus, and HiFT execution.",
"category": "tts",
"status": "community",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"zh",
"en"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"options": {
"request": [
{
"name": "reference_text",
"type": "string",
"description": "Transcript matching the reference voice audio; required by GLM-TTS zero-shot synthesis.",
"required": true
},
{
"name": "max_tokens",
"type": "int",
"description": "Maximum generated speech tokens; otherwise the official 2x-to-20x text-token bounds are used.",
"required": false,
"min": 0,
"default": 0
},
{
"name": "temperature",
"type": "float",
"description": "Speech-token temperature; official default 1.0.",
"required": false,
"min": 0.0,
"default": 1.0
},
{
"name": "top_k",
"type": "int",
"description": "Speech-token top-k; official default 25.",
"required": false,
"min": 0,
"default": 25
},
{
"name": "top_p",
"type": "float",
"description": "Speech-token nucleus threshold; official default 0.8.",
"required": false,
"min": 0.0,
"max": 1.0,
"default": 0.8
},
{
"name": "seed",
"type": "int",
"description": "Speech-token, Flow-noise, and HiFT seed.",
"required": false,
"min": 0,
"default": 0
},
{
"name": "num_inference_steps",
"type": "int",
"description": "Flow Euler steps; defaults to model config, usually official default 10.",
"required": false,
"min": 1
},
{
"name": "flow_guidance_scale",
"type": "float",
"description": "Flow classifier-free guidance rate; defaults to model config, usually official default 0.7.",
"required": false,
"min": 0.0
},
{
"name": "flow_noise_path",
"type": "path",
"description": "Optional raw float32 initial Flow noise for parity tests.",
"required": false
},
{
"name": "hift_source_random_path",
"type": "path",
"description": "Optional raw float32 HiFT phase-uniform and Gaussian values for parity tests.",
"required": false
},
{
"name": "hift_prior_noise_count",
"type": "int",
"description": "Torch RNG value offset used before HiFT source generation; default 0.",
"required": false,
"min": 0,
"default": 0
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Requested component weight storage type; default native.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "mem_saver",
"type": "bool",
"description": "Release reference-only encoders after caching the voice while keeping the generation path warm; default false.",
"required": false,
"default": false
},
{
"name": "aggressive_mem_saver",
"type": "bool",
"description": "Also release Llama, Flow, and HiFT after every stage. Minimizes VRAM but reloads the generation path on every request; default false.",
"required": false,
"default": false
},
{
"name": "reference_cache_slots",
"type": "int",
"description": "Prepared reference-audio cache slots; default 1. Use 0 to disable.",
"required": false,
"min": 0,
"default": 1
},
{
"name": "llama_weight_context_mb",
"type": "int",
"description": "Llama weight metadata context in MiB; default 8192.",
"required": false,
"min": 1,
"default": 8192
},
{
"name": "constant_context_mb",
"type": "int",
"description": "Llama constant tensor context in MiB; default 256.",
"required": false,
"min": 1,
"default": 256
}
],
"load": []
},
"runtime": {
"tags": [
"gguf"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"llama_config": "model:llm/config.json",
"llama_generation_config": "model:llm/generation_config.json",
"speech_tokenizer_config": "model:speech_tokenizer/config.json",
"speech_tokenizer_preprocessor": "model:speech_tokenizer/preprocessor_config.json",
"tokenizer_config": "model:vq32k-phoneme-tokenizer/tokenizer_config.json",
"tokenizer_vocab": "model:vq32k-phoneme-tokenizer/tokenizer_vocab.json",
"tokenizer_merges": "model:vq32k-phoneme-tokenizer/tokenizer_merges.txt",
"flow_config": "model:flow/config.yaml",
"audio_cpp_config": "model:audio_cpp_config.json"
},
"tensors": {
"llama_weights": {
"source": "weights:",
"prefix": "llama_weights"
},
"speech_tokenizer_weights": {
"source": "weights:",
"prefix": "speech_tokenizer_weights"
},
"flow_weights": {
"source": "weights:",
"prefix": "flow_weights"
},
"hift_weights": {
"source": "weights:",
"prefix": "hift_weights"
},
"campplus_weights": {
"source": "weights:",
"prefix": "campplus_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"llama_config": "model:llm/config.json",
"llama_generation_config": "model:llm/generation_config.json",
"speech_tokenizer_config": "model:speech_tokenizer/config.json",
"speech_tokenizer_preprocessor": "model:speech_tokenizer/preprocessor_config.json",
"tokenizer_config": "model:vq32k-phoneme-tokenizer/tokenizer_config.json",
"tokenizer_vocab": "model:vq32k-phoneme-tokenizer/tokenizer_vocab.json",
"tokenizer_merges": "model:vq32k-phoneme-tokenizer/tokenizer_merges.txt",
"flow_config": "model:flow/config.yaml",
"audio_cpp_config": "model:audio_cpp_config.json"
},
"tensors": {
"llama_weights": "model:llm/model.safetensors.index.json",
"speech_tokenizer_weights": "model:speech_tokenizer/model.safetensors",
"flow_weights": "model:flow/model.safetensors",
"hift_weights": "model:hift/model.safetensors",
"campplus_weights": "model:frontend/campplus.safetensors"
}
}
],
"packages": [
{
"id": "glm_tts_q8_0",
"display_name": "GLM-TTS mixed Q8_0/F16 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "GLM-TTS-Q8",
"files": [
"Text to audio (TTS)/GLM-TTS_Q8.gguf"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "mirek190/audio.cpp"
}
}
],
"dependencies": [],
"ui": {
"recommended_package": "glm_tts_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/community_models/glm_tts.md",
"docs/reports/glm_tts_validation.md",
"docs/gguf.md"
]
}
}
@@ -0,0 +1,212 @@
{
"schema_version": 1,
"family": "higgs_audio_stt",
"display_name": "Higgs Audio v3 STT",
"description": "Boson AI English speech-to-text model combining a Whisper Large v3 speech encoder with a Qwen decoder for robust ASR.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"en"
],
"capabilities": {},
"dependencies": [],
"options": {
"request": [
{
"name": "language",
"type": "string",
"description": "Transcript language code metadata; English is used when omitted.",
"required": false
},
{
"name": "max_tokens",
"type": "int",
"description": "Maximum generated transcript tokens; default 1024.",
"required": false,
"min": 1,
"default": 1024
},
{
"name": "enable_thinking",
"type": "bool",
"description": "Enable the model thinking prompt; default true.",
"required": false,
"default": true
},
{
"name": "audio_chunk_mode",
"type": "enum",
"description": "Audio chunking mode; default auto uses fixed chunks.",
"values": [
"auto",
"fixed",
"none"
],
"required": false,
"default": "auto"
},
{
"name": "audio_chunk_duration_sec",
"type": "float",
"description": "Fixed audio chunk duration in seconds; must be positive when set; default 4.",
"required": false,
"min": 0.0,
"default": 4.0
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Shared text decoder weight storage type; default native.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "audio_encoder_weight_type",
"type": "enum",
"description": "Audio encoder convolution weight storage type; default native.",
"preset": "weight_type_conv",
"required": false,
"default": "native"
},
{
"name": "text_decoder_weight_type",
"type": "enum",
"description": "Text decoder matmul weight storage type; defaults to weight_type when set, otherwise native.",
"preset": "weight_type_full",
"required": false
},
{
"name": "audio_encoder_graph_arena_mb",
"type": "int",
"description": "Audio encoder graph arena size in MiB; default 512.",
"required": false,
"min": 0,
"default": 512
},
{
"name": "text_decoder_prefill_graph_arena_mb",
"type": "int",
"description": "Text decoder prefill graph arena size in MiB; default 512.",
"required": false,
"min": 0,
"default": 512
},
{
"name": "text_decoder_decode_graph_arena_mb",
"type": "int",
"description": "Text decoder cached-step graph arena size in MiB; default 256.",
"required": false,
"min": 0,
"default": 256
},
{
"name": "text_decoder_weight_context_mb",
"type": "int",
"description": "Text decoder weight context arena size in MiB; default 4096.",
"required": false,
"min": 0,
"default": 4096
}
],
"load": []
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "higgs_audio_stt_q8_0",
"tags": [
"ASR",
"GGUF",
"Stream"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "higgs_audio_stt_q8_0",
"display_name": "Higgs Audio v3 STT Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Higgs-Audio-v3-STT-GGUF",
"files": [
"Higgs-Audio-v3-STT-GGUF/higgs-audio-v3-stt-q8_0.gguf"
],
"strip_prefix": "Higgs-Audio-v3-STT-GGUF"
},
{
"id": "higgs_audio_stt_f16",
"display_name": "Higgs Audio v3 STT F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Higgs-Audio-v3-STT-GGUF",
"files": [
"Higgs-Audio-v3-STT-GGUF/higgs-audio-v3-stt-f16.gguf"
],
"strip_prefix": "Higgs-Audio-v3-STT-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"preprocessor_config": "model:preprocessor_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": ".",
"whisper": "../whisper-large-v3"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"preprocessor_config": "whisper:preprocessor_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt"
},
"tensors": {
"weights": "model:model.safetensors.index.json"
}
}
]
}
@@ -0,0 +1,105 @@
{
"family": "higgs_audio_tts",
"display_name": "Higgs Audio v3 TTS",
"description": "Boson AI conversational TTS model for expressive speech across 100+ languages, zero-shot voice cloning, and inline control over emotion, style, prosody, pauses, and sound effects.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"100+ languages"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "higgs_audio_tts_4b_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "higgs_audio_tts_4b_q8_0",
"display_name": "Higgs Audio v3 TTS 4B Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Higgs-Audio-v3-TTS-4B-GGUF",
"files": [
"Higgs-Audio-v3-TTS-4B-GGUF/higgs-audio-v3-tts-4b-q8_0.gguf"
],
"strip_prefix": "Higgs-Audio-v3-TTS-4B-GGUF"
},
{
"id": "higgs_audio_tts_4b_bf16",
"display_name": "Higgs Audio v3 TTS 4B BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "Higgs-Audio-v3-TTS-4B-GGUF",
"files": [
"Higgs-Audio-v3-TTS-4B-GGUF/higgs-audio-v3-tts-4b-bf16.gguf"
],
"strip_prefix": "Higgs-Audio-v3-TTS-4B-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"chat_template": "model:chat_template.jinja"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"chat_template": "model:chat_template.jinja"
},
"tensors": {
"weights": "model:model.safetensors.index.json"
}
}
]
}
+265
View File
@@ -0,0 +1,265 @@
{
"family": "hviske_asr",
"schema_version": 1,
"display_name": "Hviske ASR",
"description": "Danish-optimized Conformer encoder-decoder ASR model fine-tuned from the Hviske v5 family, with selectable Cohere ASR language prompts.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline"
],
"languages": [
"ar",
"da",
"de",
"el",
"en",
"es",
"fr",
"it",
"ja",
"ko",
"nl",
"pl",
"pt",
"vi",
"zh"
],
"capabilities": {},
"dependencies": [],
"options": {
"request": [
{
"name": "language",
"type": "string",
"description": "ASR language code; default da.",
"required": false,
"default": "da"
},
{
"name": "punctuation",
"type": "bool",
"description": "Enable or disable punctuation tokens in the decoder prompt; default true.",
"required": false,
"default": true
},
{
"name": "max_tokens",
"type": "int",
"description": "Maximum generated transcript tokens; defaults to the model config.",
"required": false,
"min": 1
},
{
"name": "num_beams",
"type": "int",
"description": "Beam-search beam count; default 1 uses greedy or sampling decode.",
"required": false,
"min": 1,
"default": 1
},
{
"name": "length_penalty",
"type": "float",
"description": "Beam-search length penalty; must be positive when set; default 1.0.",
"required": false,
"min": 0.000001,
"default": 1.0
},
{
"name": "do_sample",
"type": "bool",
"description": "Enable sampling instead of greedy decode when num_beams is 1; default false.",
"required": false,
"default": false
},
{
"name": "temperature",
"type": "float",
"description": "Decoder sampling temperature; must be positive when set; default 1.0.",
"required": false,
"min": 0.000001,
"default": 1.0
},
{
"name": "top_k",
"type": "int",
"description": "Top-k sampling limit; default 50, 0 disables top-k.",
"required": false,
"min": 0,
"default": 50
},
{
"name": "top_p",
"type": "float",
"description": "Nucleus sampling limit in (0, 1]; default 1.0.",
"required": false,
"min": 0.000001,
"max": 1.0,
"default": 1.0
},
{
"name": "seed",
"type": "int",
"description": "Decoder sampling seed; random if omitted.",
"required": false,
"min": 0
},
{
"name": "audio_chunk_mode",
"type": "enum",
"description": "Audio chunking mode; default auto uses quiet-energy splitting only when audio exceeds the model clip window.",
"values": [
"auto",
"fixed",
"quiet_energy",
"none"
],
"required": false,
"default": "auto"
},
{
"name": "audio_chunk_duration_sec",
"type": "float",
"description": "Maximum audio chunk duration in seconds; defaults to the model clip window.",
"required": false,
"min": 0.000001
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Matmul weight storage type; default native.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "conv_weight_type",
"type": "enum",
"description": "Convolution weight storage type; defaults to weight_type when set, otherwise native.",
"preset": "weight_type_conv",
"required": false
},
{
"name": "weight_context_mb",
"type": "int",
"description": "Weight descriptor context size in MiB; default 32.",
"required": false,
"min": 1,
"default": 32
},
{
"name": "encoder_graph_arena_mb",
"type": "int",
"description": "Encoder graph arena size in MiB; default 512.",
"required": false,
"min": 1,
"default": 512
},
{
"name": "decoder_prefill_graph_arena_mb",
"type": "int",
"description": "Decoder prefill graph arena size in MiB; default 512.",
"required": false,
"min": 1,
"default": 512
},
{
"name": "decoder_decode_graph_arena_mb",
"type": "int",
"description": "Decoder cached-step graph arena size in MiB; default 512.",
"required": false,
"min": 1,
"default": 512
}
],
"load": []
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "hviske_asr_q8_0",
"tags": [
"ASR",
"GGUF"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
]
},
"packages": [
{
"id": "hviske_asr_q8_0",
"display_name": "Hviske v5.3 Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Hviske-v5.3-GGUF",
"files": [
"Audio to text (ASR)/hviske-v5.3_Q8.gguf"
],
"strip_prefix": "Audio to text (ASR)",
"download": {
"kind": "huggingface_snapshot",
"repo": "mirek190/audio.cpp"
}
},
{
"id": "hviske_asr_safetensors",
"display_name": "Hviske v5.3 Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "hviske-v5.3",
"files": [
"config.json",
"generation_config.json",
"model.safetensors",
"tokenizer.model"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "syvai/hviske-v5.3"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer": "model:tokenizer.model"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer": "model:tokenizer.model"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
+199
View File
@@ -0,0 +1,199 @@
{
"family": "index_tts2",
"display_name": "IndexTTS2",
"description": "Zero-shot TTS system for Chinese and English speech synthesis with voice cloning, emotion-speaker decoupling, text or audio emotion control, and explicit duration control.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"zh",
"en"
],
"capabilities": {
"tts": [
"emotion_control"
],
"clone": [
"speaker_reference",
"emotion_control"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "index_tts2_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "index_tts2_q8_0",
"display_name": "IndexTTS2 Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "IndexTTS2-GGUF",
"files": [
"IndexTTS2-GGUF/index-tts2-q8_0.gguf"
],
"strip_prefix": "IndexTTS2-GGUF"
},
{
"id": "index_tts2_f16",
"display_name": "IndexTTS2 F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "IndexTTS2-GGUF",
"files": [
"IndexTTS2-GGUF/index-tts2-f16.gguf"
],
"strip_prefix": "IndexTTS2-GGUF"
},
{
"id": "index_tts2_orig",
"display_name": "IndexTTS2 Original-Dtype GGUF",
"format": "gguf",
"precision": "orig",
"target_directory": "IndexTTS2-GGUF",
"files": [
"IndexTTS2-GGUF/index-tts2-orig.gguf"
],
"strip_prefix": "IndexTTS2-GGUF"
},
{
"id": "index_tts2_safetensors",
"display_name": "IndexTTS2 Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "IndexTTS-2",
"files": [
"config.yaml",
"bpe.model",
"gpt.safetensors"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "mlx-community/index-tts2-mlx"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.yaml",
"bpe": "model:bpe.model",
"wav2vec2bert_config": "model:w2v-bert-2.0/config.json",
"wav2vec2bert_preprocessor_config": "model:w2v-bert-2.0/preprocessor_config.json",
"bigvgan_config": "model:bigvgan/config.json",
"qwen_emotion_config": "model:qwen0.6bemo4-merge/config.json",
"qwen_emotion_generation_config": "model:qwen0.6bemo4-merge/generation_config.json",
"qwen_emotion_tokenizer": "model:qwen0.6bemo4-merge/tokenizer.json",
"qwen_emotion_tokenizer_config": "model:qwen0.6bemo4-merge/tokenizer_config.json",
"qwen_emotion_vocab": "model:qwen0.6bemo4-merge/vocab.json",
"qwen_emotion_merges": "model:qwen0.6bemo4-merge/merges.txt"
},
"tensors": {
"gpt": {
"source": "weights:",
"prefix": "gpt"
},
"s2mel": {
"source": "weights:",
"prefix": "s2mel"
},
"speaker_matrix": {
"source": "weights:",
"prefix": "speaker_matrix"
},
"emotion_matrix": {
"source": "weights:",
"prefix": "emotion_matrix"
},
"wav2vec2bert_stats": {
"source": "weights:",
"prefix": "wav2vec2bert_stats"
},
"wav2vec2bert": {
"source": "weights:",
"prefix": "wav2vec2bert"
},
"semantic_codec": {
"source": "weights:",
"prefix": "semantic_codec"
},
"campplus": {
"source": "weights:",
"prefix": "campplus"
},
"bigvgan": {
"source": "weights:",
"prefix": "bigvgan"
},
"qwen_emotion": {
"source": "weights:",
"prefix": "qwen_emotion"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.yaml",
"bpe": "model:bpe.model",
"wav2vec2bert_config": "model:w2v-bert-2.0/config.json",
"wav2vec2bert_preprocessor_config": "model:w2v-bert-2.0/preprocessor_config.json",
"bigvgan_config": "model:bigvgan/config.json",
"qwen_emotion_config": "model:qwen0.6bemo4-merge/config.json",
"qwen_emotion_generation_config": "model:qwen0.6bemo4-merge/generation_config.json",
"qwen_emotion_tokenizer": "model:qwen0.6bemo4-merge/tokenizer.json",
"qwen_emotion_tokenizer_config": "model:qwen0.6bemo4-merge/tokenizer_config.json",
"qwen_emotion_vocab": "model:qwen0.6bemo4-merge/vocab.json",
"qwen_emotion_merges": "model:qwen0.6bemo4-merge/merges.txt"
},
"tensors": {
"gpt": "model:gpt.safetensors",
"s2mel": "model:s2mel.safetensors",
"speaker_matrix": "model:feat1.safetensors",
"emotion_matrix": "model:feat2.safetensors",
"wav2vec2bert_stats": "model:wav2vec2bert_stats.safetensors",
"wav2vec2bert": "model:w2v-bert-2.0/model.safetensors",
"semantic_codec": "model:semantic_codec_model.safetensors",
"campplus": "model:campplus.safetensors",
"bigvgan": "model:bigvgan/model.safetensors",
"qwen_emotion": "model:qwen0.6bemo4-merge/model.safetensors"
}
}
]
}
+155
View File
@@ -0,0 +1,155 @@
{
"schema_version": 1,
"family": "inflect_v2",
"display_name": "Inflect Micro v2",
"description": "Compact English VITS text-to-speech models with a native GGML inference path and an external eSpeak-ng phonemizer.",
"category": "tts",
"status": "community",
"tasks": [
"tts"
],
"modes": [
"offline"
],
"languages": [
"en"
],
"runtime": {
"tags": [
"gguf"
]
},
"capabilities": {
"tts": [
"long_form"
]
},
"options": {
"request": [
{
"name": "speaking_rate",
"type": "float",
"description": "Speech speed multiplier mapped directly to Inflect speed; default 1.0.",
"required": false,
"min": 0.5,
"max": 2.0,
"default": 1.0
},
{
"name": "variation",
"type": "float",
"description": "Latent Gaussian variation; default 0.667.",
"required": false,
"min": 0.0,
"max": 1.0,
"default": 0.667
},
{
"name": "seed",
"type": "int",
"description": "Non-negative latent noise seed; default 0.",
"required": false,
"min": 0,
"default": 0
},
{
"name": "text_chunk_mode",
"type": "enum",
"description": "Long-form text chunking mode.",
"values": [
"word_budget"
],
"required": false,
"default": "word_budget"
},
{
"name": "text_chunk_size",
"type": "int",
"description": "Maximum Unicode codepoints per long-form text chunk; default 280.",
"required": false,
"min": 1,
"default": 280
}
],
"session": [
{
"name": "espeak_library_path",
"type": "path",
"description": "Optional explicit path to the eSpeak-ng shared library.",
"required": false
},
{
"name": "espeak_data_path",
"type": "path",
"description": "Optional explicit path to the directory containing espeak-ng-data.",
"required": false
}
],
"load": []
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "inflect_micro_v2_orig",
"display_name": "Inflect Micro v2 Original-Dtype GGUF",
"default": true,
"format": "gguf",
"precision": "orig",
"target_directory": "Inflect-Micro-v2-GGUF",
"files": [
"Inflect-Micro-v2-GGUF/inflect-micro-v2-orig.gguf"
],
"strip_prefix": "Inflect-Micro-v2-GGUF"
}
],
"dependencies": [],
"ui": {
"recommended_package": "inflect_micro_v2_orig",
"tags": [
"TTS",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/community_models/inflect_v2.md",
"docs/gguf.md"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json"
},
"tensors": {
"weights": {
"source": "weights:",
"prefix": "weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
@@ -0,0 +1,406 @@
{
"schema_version": 1,
"family": "irodori_tts",
"display_name": "Irodori-TTS",
"description": "Japanese TTS model based on RF-DiT continuous audio latents, supporting zero-shot voice cloning, automatic duration prediction, multimodal voice design, and emoji-style control.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone",
"design"
],
"modes": [
"offline"
],
"languages": [
"ja"
],
"capabilities": {
"clone": [
"speaker_reference"
],
"design": [
"voice_design"
]
},
"options": {
"request": [
{
"name": "language",
"type": "string",
"description": "Text language code; Irodori-TTS supports Japanese only.",
"required": false,
"default": "ja"
},
{
"name": "caption",
"type": "string",
"description": "VoiceDesign caption describing target voice identity, style, or emotion; supported only by caption-conditioned checkpoints.",
"required": false
},
{
"name": "no_ref",
"type": "bool",
"description": "Use no-reference generation; default true unless a speaker reference is provided.",
"required": false,
"default": true
},
{
"name": "num_inference_steps",
"type": "int",
"description": "RF diffusion steps; default 40.",
"required": false,
"min": 1,
"default": 40
},
{
"name": "duration_sec",
"type": "float",
"description": "Explicit output duration in seconds; must be positive when set, otherwise predicted duration is used.",
"required": false,
"min": 0.0
},
{
"name": "duration_scale",
"type": "float",
"description": "Predicted-duration multiplier; must be positive when set; default 1.0.",
"required": false,
"min": 0.0,
"default": 1.0
},
{
"name": "text_chunk_mode",
"type": "enum",
"description": "Text chunking mode; default endline.",
"values": [
"japanese",
"endline"
],
"required": false,
"default": "endline"
},
{
"name": "text_chunk_size",
"type": "int",
"description": "Maximum characters per text chunk; default uses the model text-token window.",
"required": false,
"min": 1
},
{
"name": "min_duration_sec",
"type": "float",
"description": "Minimum generated duration in seconds; must be positive and no greater than max_duration_sec; default 0.5.",
"required": false,
"min": 0.0,
"default": 0.5
},
{
"name": "max_duration_sec",
"type": "float",
"description": "Maximum generated duration in seconds; must be at least min_duration_sec; default 30.",
"required": false,
"min": 0.0,
"default": 30.0
},
{
"name": "text_guidance_scale",
"type": "float",
"description": "Text classifier-free guidance scale; default 3.0.",
"required": false,
"min": 0.0,
"default": 3.0
},
{
"name": "speaker_guidance_scale",
"type": "float",
"description": "Speaker classifier-free guidance scale; default 5.0.",
"required": false,
"min": 0.0,
"default": 5.0
},
{
"name": "caption_guidance_scale",
"type": "float",
"description": "Caption classifier-free guidance scale; default 3.0.",
"required": false,
"min": 0.0,
"default": 3.0
},
{
"name": "guidance_scale",
"type": "float",
"description": "Override all classifier-free guidance scales when set.",
"required": false,
"min": 0.0
},
{
"name": "guidance_mode",
"type": "enum",
"description": "Classifier-free guidance combination mode; default independent.",
"values": [
"independent",
"joint",
"alternating"
],
"required": false,
"default": "independent"
},
{
"name": "guidance_min_t",
"type": "float",
"description": "Minimum diffusion timestep value where guidance is active; default 0.5.",
"required": false,
"min": 0.0,
"default": 0.5
},
{
"name": "guidance_max_t",
"type": "float",
"description": "Maximum diffusion timestep value where guidance is active; default 1.0.",
"required": false,
"min": 0.0,
"default": 1.0
},
{
"name": "seed",
"type": "int",
"description": "Generation seed for reproducible output; omitted uses a random seed.",
"required": false,
"min": 0
},
{
"name": "trim_tail",
"type": "bool",
"description": "Trim trailing silence-like samples; default true.",
"required": false,
"default": true
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Model weight storage type; default native.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "codec_weight_type",
"type": "enum",
"description": "DACVAE codec weight storage type; default native.",
"preset": "weight_type_codec_q8",
"required": false,
"default": "native"
},
{
"name": "condition_graph_arena_mb",
"type": "int",
"description": "Condition encoder graph arena size in MiB; default 256.",
"required": false,
"min": 0,
"default": 256
},
{
"name": "rf_graph_arena_mb",
"type": "int",
"description": "RF sampler graph arena size in MiB; default 768.",
"required": false,
"min": 0,
"default": 768
},
{
"name": "codec_graph_arena_mb",
"type": "int",
"description": "DACVAE codec graph arena size in MiB; default 512.",
"required": false,
"min": 0,
"default": 512
},
{
"name": "condition_weight_context_mb",
"type": "int",
"description": "Condition encoder weight context size in MiB; default 32.",
"required": false,
"min": 0,
"default": 32
},
{
"name": "rf_weight_context_mb",
"type": "int",
"description": "RF sampler weight context size in MiB; default 32.",
"required": false,
"min": 0,
"default": 32
},
{
"name": "codec_weight_context_mb",
"type": "int",
"description": "DACVAE codec weight context size in MiB; default 32.",
"required": false,
"min": 0,
"default": 32
},
{
"name": "mem_saver",
"type": "bool",
"description": "Release staged runtime graphs after request phases; default true.",
"required": false,
"default": true
},
{
"name": "reference_cache_slots",
"type": "int",
"description": "Prepared reference-speaker cache slots; default 1.",
"required": false,
"min": 0,
"default": 1
}
],
"load": []
},
"dependencies": [],
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "irodori_tts_v4_small_q8_0",
"tags": [
"TTS",
"Clone",
"Design",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "irodori_tts_v4_small_q8_0",
"display_name": "Irodori-TTS v4 Small Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Irodori-TTS-v4-Small-GGUF",
"files": [
"Irodori-TTS-v4-Small-GGUF/irodori-tts-v4-small-q8_0.gguf"
],
"strip_prefix": "Irodori-TTS-v4-Small-GGUF"
},
{
"id": "irodori_tts_v4_small_f16",
"display_name": "Irodori-TTS v4 Small F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Irodori-TTS-v4-Small-GGUF",
"files": [
"Irodori-TTS-v4-Small-GGUF/irodori-tts-v4-small-f16.gguf"
],
"strip_prefix": "Irodori-TTS-v4-Small-GGUF"
},
{
"id": "irodori_tts_600m_v3_voicedesign_q8_0",
"display_name": "Irodori-TTS 600M v3 VoiceDesign Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "Irodori-TTS-600M-v3-VoiceDesign-GGUF",
"files": [
"Irodori-TTS-600M-v3-VoiceDesign-GGUF/irodori-tts-600m-v3-voicedesign-q8_0.gguf"
],
"strip_prefix": "Irodori-TTS-600M-v3-VoiceDesign-GGUF"
},
{
"id": "irodori_tts_600m_v3_voicedesign_f16",
"display_name": "Irodori-TTS 600M v3 VoiceDesign F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Irodori-TTS-600M-v3-VoiceDesign-GGUF",
"files": [
"Irodori-TTS-600M-v3-VoiceDesign-GGUF/irodori-tts-600m-v3-voicedesign-f16.gguf"
],
"strip_prefix": "Irodori-TTS-600M-v3-VoiceDesign-GGUF"
},
{
"id": "irodori_tts_500m_v3_q8_0",
"display_name": "Irodori-TTS 500M v3 Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "Irodori-TTS-500M-v3-GGUF",
"files": [
"Irodori-TTS-500M-v3-GGUF/irodori-tts-500m-v3-q8_0.gguf"
],
"strip_prefix": "Irodori-TTS-500M-v3-GGUF"
},
{
"id": "irodori_tts_500m_v3_f16",
"display_name": "Irodori-TTS 500M v3 F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Irodori-TTS-500M-v3-GGUF",
"files": [
"Irodori-TTS-500M-v3-GGUF/irodori-tts-500m-v3-f16.gguf"
],
"strip_prefix": "Irodori-TTS-500M-v3-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"model_config": "model:model_config.json"
},
"optional_files": {
"tokenizer_json": "model:tokenizer.json",
"tokenizer_v4_json": "model:tokenizer/tokenizer.json",
"pretrained_text_config": "model:text_encoder_config.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"codec_weights": {
"source": "weights:",
"prefix": "codec_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": ".",
"tokenizer": "../llm-jp-3-150m",
"codec": "../Semantic-DACVAE-Japanese-32dim"
},
"files": {
"model_config": "model:model_config.json"
},
"optional_files": {
"tokenizer_json": "tokenizer:tokenizer.json",
"tokenizer_v4_json": "model:tokenizer/tokenizer.json",
"pretrained_text_config": "model:text_encoder_config.json"
},
"tensors": {
"model_weights": "model:model.safetensors",
"codec_weights": "codec:weights.safetensors"
}
}
]
}
+190
View File
@@ -0,0 +1,190 @@
{
"schema_version": 1,
"family": "kroko_asr",
"display_name": "Kroko Community ASR",
"description": "Native Zipformer2 RNN-T transcription for the public free Kroko Community single-language packages, with offline and stateful streaming inference, greedy and modified beam decoding, hotwords, endpoint segments, partial results, and word timestamps.",
"category": "asr",
"status": "community",
"tasks": [
"asr"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"de",
"en",
"es",
"fr",
"it",
"he",
"nl",
"pt",
"sv",
"tr"
],
"capabilities": {
"asr": [
"word_timestamps",
"partial_results",
"segments"
]
},
"options": {
"request": [
{
"name": "language",
"type": "string",
"description": "Language code matching the selected single-language Kroko package; default auto uses the package language.",
"required": false,
"default": "auto"
},
{
"name": "decoding_method",
"type": "enum",
"description": "RNN-T decoding method; greedy search remains the parity-tested default.",
"values": [
"greedy_search",
"modified_beam_search"
],
"required": false,
"default": "greedy_search"
},
{
"name": "num_beams",
"type": "int",
"description": "Maximum active hypotheses retained by modified beam search.",
"required": false,
"min": 1,
"max": 64,
"default": 4
},
{
"name": "blank_penalty",
"type": "float",
"description": "Non-negative score subtracted from the RNN-T blank logit.",
"required": false,
"min": 0.0,
"default": 0.0
},
{
"name": "hotwords",
"type": "string",
"description": "Slash- or newline-separated natural-text phrases; requires modified beam search.",
"required": false
},
{
"name": "hotwords_score",
"type": "float",
"description": "Non-negative per-token context boost for hotword phrases.",
"required": false,
"min": 0.0,
"default": 1.5
},
{
"name": "enable_endpoint",
"type": "bool",
"description": "Enable automatic endpoint speech segments.",
"required": false,
"default": false
},
{
"name": "rule1_min_trailing_silence_sec",
"type": "float",
"description": "Endpoint timeout in seconds even when no speech token was decoded.",
"required": false,
"min": 0.0,
"default": 2.4
},
{
"name": "rule2_min_trailing_silence_sec",
"type": "float",
"description": "Endpoint silence in seconds after a speech token was decoded.",
"required": false,
"min": 0.0,
"default": 1.2
},
{
"name": "rule3_min_utterance_length_sec",
"type": "float",
"description": "Maximum utterance duration in seconds before an endpoint.",
"required": false,
"min": 0.0,
"default": 20.0
}
],
"session": [],
"load": []
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokens": "model:tokens.txt"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokens": "model:tokens.txt"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
],
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "kroko_asr_community_q8_0",
"display_name": "Kroko Community ASR Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Kroko-ASR-GGUF",
"files": [
"Kroko-ASR-GGUF/kroko-en-community-64-l-q8_0.gguf"
],
"strip_prefix": "Kroko-ASR-GGUF"
}
],
"dependencies": [],
"ui": {
"recommended_package": "kroko_asr_community_q8_0",
"tags": [
"ASR",
"GGUF"
],
"docs": [
"docs/asr.md",
"docs/community_models/kroko_asr.md",
"docs/gguf.md"
]
}
}
+125
View File
@@ -0,0 +1,125 @@
{
"family": "miotts",
"display_name": "MioTTS",
"description": "Lightweight LLM-based English and Japanese TTS family built on MioCodec, supporting low-latency speech generation and zero-shot voice cloning from short reference audio.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"en",
"ja"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "miotts_1_7b_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "miotts_1_7b_q8_0",
"display_name": "MioTTS 1.7B Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "MioTTS-1.7B-GGUF",
"files": [
"MioTTS-1.7B-GGUF/miotts-1.7b-q8_0.gguf"
],
"strip_prefix": "MioTTS-1.7B-GGUF"
},
{
"id": "miotts_1_7b_bf16",
"display_name": "MioTTS 1.7B BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "MioTTS-1.7B-GGUF",
"files": [
"MioTTS-1.7B-GGUF/miotts-1.7b-bf16.gguf"
],
"strip_prefix": "MioTTS-1.7B-GGUF"
},
{
"id": "miotts_1_7b_orig",
"display_name": "MioTTS 1.7B Original-Dtype GGUF",
"format": "gguf",
"precision": "orig",
"target_directory": "MioTTS-1.7B-GGUF",
"files": [
"MioTTS-1.7B-GGUF/miotts-1.7b-orig.gguf"
],
"strip_prefix": "MioTTS-1.7B-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt"
},
"optional_files": {
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt"
},
"optional_files": {
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
@@ -0,0 +1,149 @@
{
"family": "moss_tts_local",
"display_name": "MOSS-TTS-Local",
"description": "Flagship MOSS-TTS model for high-fidelity 31-language and code-switched speech, zero-shot voice cloning, long-form generation, and fine-grained Pinyin, phoneme, and duration control.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"ar",
"cs",
"da",
"de",
"el",
"en",
"es",
"fa",
"fi",
"fr",
"he",
"hi",
"hu",
"it",
"ja",
"ko",
"mk",
"ms",
"nl",
"pl",
"pt",
"ro",
"ru",
"sv",
"sw",
"th",
"tl",
"tr",
"vi",
"yue",
"zh"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "moss_tts_local_v1_5_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/models/moss_tts.md",
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "moss_tts_local_v1_5_q8_0",
"display_name": "MOSS-TTS-Local v1.5 Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "MOSS-TTS-Local-v1.5-GGUF",
"files": [
"MOSS-TTS-Local-v1.5-GGUF/moss-tts-local-v1.5-q8_0.gguf"
],
"strip_prefix": "MOSS-TTS-Local-v1.5-GGUF"
},
{
"id": "moss_tts_local_v1_5_bf16",
"display_name": "MOSS-TTS-Local v1.5 BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "MOSS-TTS-Local-v1.5-GGUF",
"files": [
"MOSS-TTS-Local-v1.5-GGUF/moss-tts-local-v1.5-bf16.gguf"
],
"strip_prefix": "MOSS-TTS-Local-v1.5-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_vocab": "model:vocab.json",
"tokenizer_merges": "model:merges.txt",
"audio_tokenizer_config": "model:audio_tokenizer/config.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"audio_tokenizer_weights": {
"source": "weights:",
"prefix": "audio_tokenizer_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": ".",
"audio_tokenizer": "audio_tokenizer"
},
"files": {
"config": "model:config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_vocab": "model:vocab.json",
"tokenizer_merges": "model:merges.txt",
"audio_tokenizer_config": "audio_tokenizer:config.json"
},
"tensors": {
"model_weights": "model:model.safetensors",
"audio_tokenizer_weights": "audio_tokenizer:model.safetensors.index.json"
}
}
]
}
@@ -0,0 +1,133 @@
{
"family": "moss_tts_nano",
"display_name": "MOSS-TTS-Nano",
"description": "Compact deployment-first MOSS-TTS model for real-time multilingual speech generation, lightweight integration, and zero-shot voice cloning.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"ar",
"cs",
"da",
"de",
"el",
"en",
"es",
"fa",
"fr",
"hu",
"it",
"ja",
"ko",
"pl",
"pt",
"ru",
"sv",
"tr",
"zh"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "moss_tts_nano_100m_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/models/moss_tts.md",
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "moss_tts_nano_100m_q8_0",
"display_name": "MOSS-TTS-Nano 100M Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "MOSS-TTS-Nano-100M-GGUF",
"files": [
"MOSS-TTS-Nano-100M-GGUF/moss-tts-nano-100m-q8_0.gguf"
],
"strip_prefix": "MOSS-TTS-Nano-100M-GGUF"
},
{
"id": "moss_tts_nano_100m_bf16",
"display_name": "MOSS-TTS-Nano 100M BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "MOSS-TTS-Nano-100M-GGUF",
"files": [
"MOSS-TTS-Nano-100M-GGUF/moss-tts-nano-100m-bf16.gguf"
],
"strip_prefix": "MOSS-TTS-Nano-100M-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_model": "model:tokenizer.model",
"audio_tokenizer_config": "model:audio_tokenizer/config.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"audio_tokenizer_weights": {
"source": "weights:",
"prefix": "audio_tokenizer_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": ".",
"audio_tokenizer": "audio_tokenizer"
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_model": "model:tokenizer.model",
"audio_tokenizer_config": "audio_tokenizer:config.json"
},
"tensors": {
"model_weights": "model:model.safetensors",
"audio_tokenizer_weights": "audio_tokenizer:model.safetensors.index.json"
}
}
]
}
@@ -0,0 +1,156 @@
{
"family": "nemotron_asr",
"display_name": "Nemotron 3.5 ASR",
"description": "NVIDIA 600M streaming ASR model for low-latency and batch transcription across 40 language-locales, with native punctuation, capitalization, automatic language detection, and configurable chunk sizes.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"ar-AR",
"bg-BG",
"cs-CZ",
"da-DK",
"de-DE",
"el-GR",
"en-GB",
"en-US",
"es-ES",
"es-US",
"et-EE",
"fi-FI",
"fr-CA",
"fr-FR",
"he-IL",
"hi-IN",
"hr-HR",
"hu-HU",
"it-IT",
"ja-JP",
"ko-KR",
"lt-LT",
"lv-LV",
"mt-MT",
"nb-NO",
"nl-NL",
"nn-NO",
"pl-PL",
"pt-BR",
"pt-PT",
"ro-RO",
"ru-RU",
"sk-SK",
"sl-SI",
"sv-SE",
"th-TH",
"tr-TR",
"uk-UA",
"vi-VN",
"zh-CN"
],
"capabilities": {},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "nemotron_asr_q8_0",
"tags": [
"ASR",
"GGUF",
"Stream"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "nemotron_asr_q8_0",
"display_name": "Nemotron 3.5 ASR Streaming 0.6B Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF",
"files": [
"Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf"
],
"strip_prefix": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF"
},
{
"id": "nemotron_asr_f16",
"display_name": "Nemotron 3.5 ASR Streaming 0.6B F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF",
"files": [
"Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-f16.gguf"
],
"strip_prefix": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF"
},
{
"id": "nemotron_asr_safetensors",
"display_name": "Nemotron 3.5 ASR Streaming 0.6B Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "nemotron-3.5-asr-streaming-0.6b",
"files": [
"config.json",
"model.safetensors",
"processor_config.json",
"tokenizer.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "nvidia/nemotron-3.5-asr-streaming-0.6b"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"processor_config": "model:processor_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"processor_config": "model:processor_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
+157
View File
@@ -0,0 +1,157 @@
{
"family": "omnivoice",
"display_name": "OmniVoice",
"description": "Massively multilingual zero-shot TTS model from k2-fsa for 600+ languages, supporting short-reference voice cloning, attribute-based voice design, pronunciation controls, and nonverbal tags.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone",
"design"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"600+ languages"
],
"capabilities": {
"clone": [
"speaker_reference"
],
"design": [
"voice_design"
]
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "omnivoice_q8_0",
"tags": [
"TTS",
"Clone",
"Design",
"GGUF",
"Stream"
],
"docs": [
"docs/models/omnivoice.md",
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "omnivoice_q8_0",
"display_name": "OmniVoice Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "OmniVoice-GGUF",
"files": [
"OmniVoice-GGUF/omnivoice-q8_0.gguf"
],
"strip_prefix": "OmniVoice-GGUF"
},
{
"id": "omnivoice_bf16",
"display_name": "OmniVoice BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "OmniVoice-GGUF",
"files": [
"OmniVoice-GGUF/omnivoice-bf16.gguf"
],
"strip_prefix": "OmniVoice-GGUF"
},
{
"id": "omnivoice_f16",
"display_name": "OmniVoice F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "OmniVoice-GGUF",
"files": [
"OmniVoice-GGUF/omnivoice-f16.gguf"
],
"strip_prefix": "OmniVoice-GGUF"
},
{
"id": "omnivoice_safetensors",
"display_name": "OmniVoice Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "OmniVoice",
"files": [
"config.json",
"model.safetensors",
"tokenizer.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "k2-fsa/OmniVoice"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"audio_tokenizer_config": "model:audio_tokenizer/config.json",
"audio_tokenizer_preprocessor": "model:audio_tokenizer/preprocessor_config.json"
},
"optional_files": {
"chat_template": "model:chat_template.jinja"
},
"tensors": {
"weights": {
"source": "weights:",
"prefix": "weights"
},
"audio_tokenizer_weights": {
"source": "weights:",
"prefix": "audio_tokenizer_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"audio_tokenizer_config": "model:audio_tokenizer/config.json",
"audio_tokenizer_preprocessor": "model:audio_tokenizer/preprocessor_config.json"
},
"optional_files": {
"chat_template": "model:chat_template.jinja"
},
"tensors": {
"weights": "model:model.safetensors",
"audio_tokenizer_weights": "model:audio_tokenizer/model.safetensors"
}
}
]
}
+300
View File
@@ -0,0 +1,300 @@
{
"schema_version": 1,
"family": "outetts",
"display_name": "Llama-OuteTTS 1.0",
"description": "Llama-based open-weight TTS model for 23-language speech synthesis with one-shot voice cloning from short reference audio and automatic word-alignment support.",
"category": "tts",
"status": "community",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"ar",
"be",
"bn",
"de",
"en",
"es",
"fa",
"fr",
"hu",
"it",
"ja",
"ka",
"ko",
"lt",
"lv",
"nl",
"pl",
"pt",
"ru",
"sw",
"ta",
"uk",
"zh"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"options": {
"request": [
{
"name": "max_tokens",
"type": "int",
"description": "Maximum generated audio tokens per chunk. When omitted, OuteTTS estimates a safe value from each chunk.",
"required": false,
"min": 1
},
{
"name": "temperature",
"type": "float",
"description": "Sampling temperature; default 0.4 for cloning, otherwise model config default.",
"required": false,
"min": 0.0
},
{
"name": "top_k",
"type": "int",
"description": "Top-k sampling; default 40 for cloning, otherwise model config default.",
"required": false,
"min": 0
},
{
"name": "top_p",
"type": "float",
"description": "Nucleus sampling in (0, 1]; default 0.9 for cloning, otherwise model config default.",
"required": false,
"min": 0.0,
"max": 1.0
},
{
"name": "min_p",
"type": "float",
"description": "Minimum probability relative to the best token; default 0.05 for cloning, otherwise model config default.",
"required": false,
"min": 0.0,
"max": 1.0
},
{
"name": "repetition_penalty",
"type": "float",
"description": "Positive windowed repetition penalty; default 1.1.",
"required": false,
"min": 0.0,
"default": 1.1
},
{
"name": "repetition_window",
"type": "int",
"description": "Recent-token penalty window; default 64.",
"required": false,
"min": 0,
"default": 64
},
{
"name": "seed",
"type": "int",
"description": "Sampling seed; cloning defaults to 4099 for native weights and 42 for quantized weights.",
"required": false,
"min": 0
},
{
"name": "reference_text",
"type": "string",
"description": "Transcript matching the reference voice audio for voice cloning.",
"required": false
},
{
"name": "reference_language",
"type": "string",
"description": "Language code used to align the reference transcript; default en.",
"required": false,
"default": "en"
},
{
"name": "text_chunk_size",
"type": "int",
"description": "Maximum UTF-8 codepoints per long-form text chunk; default 256. Chunks are split further when required by max_tokens or context budget.",
"required": false,
"min": 1,
"default": 256
},
{
"name": "text_chunk_mode",
"type": "enum",
"description": "Framework long-form text chunking mode; default word_budget.",
"preset": "text_chunk_mode_full",
"required": false,
"default": "word_budget"
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Language-model weight storage type. Quantized CUDA voice cloning is expanded to F32 in memory for generation correctness.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "llama_weight_context_mb",
"type": "int",
"description": "Language-model weight context size in MiB; default 4096.",
"required": false,
"min": 1,
"default": 4096
},
{
"name": "constant_context_mb",
"type": "int",
"description": "Language-model constant tensor context size in MiB; default 256.",
"required": false,
"min": 1,
"default": 256
},
{
"name": "dac_weight_context_mb",
"type": "int",
"description": "DAC decoder weight context size in MiB; default 1024.",
"required": false,
"min": 1,
"default": 1024
},
{
"name": "dac_graph_arena_mb",
"type": "int",
"description": "DAC decoder graph arena size in MiB; default 1536.",
"required": false,
"min": 1,
"default": 1536
},
{
"name": "aligner_path",
"type": "path",
"description": "Optional Qwen3 Forced Aligner override. Cloning automatically uses the aligner embedded in a standalone OuteTTS GGUF when present.",
"required": false
},
{
"name": "reference_cache_slots",
"type": "int",
"description": "Prepared reference-profile cache slots; default 1, set 0 to disable.",
"required": false,
"min": 0,
"default": 1
},
{
"name": "mem_saver",
"type": "bool",
"description": "Release cached-step and aligner runtime state after use; default false.",
"required": false,
"default": false
}
],
"load": []
},
"runtime": {
"tags": [
"gguf"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"special_tokens_map": "model:special_tokens_map.json",
"dac_config": "model:dac/config.json"
},
"optional_files": {
"aligner_config": "model:aligner/config.json",
"aligner_generation_config": "model:aligner/generation_config.json",
"aligner_tokenizer_config": "model:aligner/tokenizer_config.json",
"aligner_preprocessor_config": "model:aligner/preprocessor_config.json",
"aligner_processor_config": "model:aligner/processor_config.json",
"aligner_chat_template": "model:aligner/chat_template.json",
"aligner_chat_template_jinja": "model:aligner/chat_template.jinja",
"aligner_vocab": "model:aligner/vocab.json",
"aligner_merges": "model:aligner/merges.txt",
"aligner_tokenizer_json": "model:aligner/tokenizer.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"dac_weights": {
"source": "weights:",
"prefix": "dac_weights"
},
"aligner_weights": {
"source": "weights:",
"prefix": "aligner_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": ".",
"dac": "../DAC.speech.v1.0"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer": "model:tokenizer.json",
"tokenizer_config": "model:tokenizer_config.json",
"special_tokens_map": "model:special_tokens_map.json",
"dac_config": "dac:config.json"
},
"tensors": {
"model_weights": "model:model.safetensors",
"dac_weights": "dac:model.safetensors"
}
}
],
"packages": [
{
"id": "outetts_1_0_1b_q8_0",
"display_name": "Llama-OuteTTS 1.0 1B Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Llama-OuteTTS-1.0-1B_Q8",
"files": [
"Text to audio (TTS)/Llama-OuteTTS-1.0-1B_Q8.gguf"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "mirek190/audio.cpp"
}
}
],
"dependencies": [],
"ui": {
"recommended_package": "outetts_1_0_1b_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/community_models/outetts.md",
"docs/reports/outetts_validation.md",
"docs/gguf.md"
]
}
}
@@ -0,0 +1,260 @@
{
"schema_version": 1,
"family": "parakeet_tdt",
"display_name": "Parakeet-TDT 0.6B v3",
"description": "NVIDIA Parakeet-TDT 0.6B v3 FastConformer-TDT ASR covering 25 European languages with automatic language detection. Supports the upstream Transformers-compatible safetensors package and standalone audio.cpp GGUF, with offline full-context, bounded-window long-form, and buffered streaming; the checkpoint uses unlimited bidirectional attention and is not a native cache-aware streaming model.",
"category": "asr",
"status": "community",
"tasks": [
"asr"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"bg",
"cs",
"da",
"de",
"el",
"en",
"es",
"et",
"fi",
"fr",
"hr",
"hu",
"it",
"lt",
"lv",
"mt",
"nl",
"pl",
"pt",
"ro",
"ru",
"sk",
"sl",
"sv",
"uk"
],
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"capabilities": {
"asr": [
"word_timestamps",
"partial_results"
]
},
"options": {
"request": [
{
"name": "max_tokens",
"type": "int",
"description": "Maximum TDT generated tokens; 0 or omitted uses the model-derived limit.",
"required": false,
"min": 0,
"default": 0
},
{
"name": "keep_language_tags",
"type": "bool",
"description": "Keep language tag tokens in decoded text; default false.",
"required": false,
"default": false
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Shared matmul weight storage type; default native.",
"preset": "weight_type_full",
"required": false,
"default": "native"
},
{
"name": "matmul_weight_type",
"type": "enum",
"description": "Encoder and decoder matmul weight storage type; defaults to weight_type, which defaults to native. Q8_0 measured 1.79x faster on the tested CPU and changed roughly 8 percent of transcripts without moving aggregate word error rate.",
"preset": "weight_type_full",
"required": false
},
{
"name": "conv_weight_type",
"type": "enum",
"description": "Convolution weight storage type; default native.",
"preset": "weight_type_conv",
"required": false,
"default": "native"
},
{
"name": "perf_mode",
"type": "enum",
"description": "Encoder attention implementation. Default off uses the validated relative-attention path; flash_attention enables the fused implementation, which was numerically validated but slower on the tested hardware.",
"preset": "perf_mode_flash_attention",
"required": false,
"default": "off"
},
{
"name": "weight_context_mb",
"type": "int",
"description": "Weight context arena size in MiB; default 3072.",
"required": false,
"min": 1,
"default": 3072
},
{
"name": "encoder_graph_arena_mb",
"type": "int",
"description": "Encoder graph arena size in MiB; default 1024.",
"required": false,
"min": 1,
"default": 1024
},
{
"name": "decoder_graph_arena_mb",
"type": "int",
"description": "Decoder graph arena size in MiB; default 256.",
"required": false,
"min": 1,
"default": 256
},
{
"name": "audio_chunk_duration_sec",
"type": "float",
"description": "Center-region duration for buffered streaming in seconds; default 2. Fixed context windows are re-encoded rather than cache-aware.",
"required": false,
"min": 0.001,
"default": 2.0
},
{
"name": "left_context_sec",
"type": "float",
"description": "Past context included when re-encoding each buffered-streaming window in seconds; default 10.",
"required": false,
"min": 0.0,
"default": 10.0
},
{
"name": "right_context_sec",
"type": "float",
"description": "Future lookahead included when re-encoding each buffered-streaming window in seconds; default 2 and adds equivalent partial-result latency.",
"required": false,
"min": 0.0,
"default": 2.0
},
{
"name": "streaming_attention_mode",
"type": "enum",
"description": "Attention policy inside each buffered window. full_context preserves bidirectional attention over the bounded window.",
"values": [
"full_context"
],
"required": false,
"default": "full_context"
},
{
"name": "offline_mode",
"type": "enum",
"description": "Offline encoder scheduling. full_context encodes the whole utterance, long_form uses bounded overlapping windows, and auto selects long_form beyond audio_chunk_threshold_sec.",
"values": [
"full_context",
"long_form",
"auto"
],
"required": false,
"default": "full_context"
},
{
"name": "audio_chunk_threshold_sec",
"type": "float",
"description": "Duration threshold used by offline_mode=auto before switching to bounded-window long-form execution; default 30 seconds.",
"required": false,
"min": 0.001,
"default": 30.0
}
],
"load": []
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "parakeet_tdt_q8_0",
"display_name": "Parakeet-TDT 0.6B v3 Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Parakeet-TDT-0.6B-v3-GGUF",
"files": [
"Parakeet-TDT-0.6B-v3-GGUF/parakeet-tdt-0.6b-v3-q8_0.gguf"
],
"strip_prefix": "Parakeet-TDT-0.6B-v3-GGUF"
},
{
"id": "parakeet_tdt_f16",
"display_name": "Parakeet-TDT 0.6B v3 F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Parakeet-TDT-0.6B-v3-GGUF",
"files": [
"Parakeet-TDT-0.6B-v3-GGUF/parakeet-tdt-0.6b-v3-f16.gguf"
],
"strip_prefix": "Parakeet-TDT-0.6B-v3-GGUF"
}
],
"dependencies": [],
"ui": {
"recommended_package": "parakeet_tdt_q8_0",
"tags": [
"ASR",
"GGUF"
],
"docs": [
"docs/community_models/parakeet_tdt.md"
]
},
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"processor_config": "model:processor_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"processor_config": "model:processor_config.json",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
+238
View File
@@ -0,0 +1,238 @@
{
"family": "pocket_tts",
"display_name": "PocketTTS",
"description": "Kyutai 100M-parameter CPU-friendly TTS package set for real-time local synthesis and small-footprint voice cloning in English, German, Italian, Portuguese, and Spanish.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"en",
"de",
"it",
"pt",
"es"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "pocket_tts_english_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "pocket_tts_english_q8_0",
"display_name": "PocketTTS English Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "PocketTTS-GGUF/english",
"files": [
"PocketTTS-GGUF/english/pocket-tts-english-q8_0.gguf"
],
"strip_prefix": "PocketTTS-GGUF/english"
},
{
"id": "pocket_tts_english_bf16",
"display_name": "PocketTTS English BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "PocketTTS-GGUF/english",
"files": [
"PocketTTS-GGUF/english/pocket-tts-english-bf16.gguf"
],
"strip_prefix": "PocketTTS-GGUF/english"
},
{
"id": "pocket_tts_german_q8_0",
"display_name": "PocketTTS German Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "PocketTTS-GGUF/german",
"files": [
"PocketTTS-GGUF/german/pocket-tts-german-q8_0.gguf"
],
"strip_prefix": "PocketTTS-GGUF/german"
},
{
"id": "pocket_tts_german_bf16",
"display_name": "PocketTTS German BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "PocketTTS-GGUF/german",
"files": [
"PocketTTS-GGUF/german/pocket-tts-german-bf16.gguf"
],
"strip_prefix": "PocketTTS-GGUF/german"
},
{
"id": "pocket_tts_italian_q8_0",
"display_name": "PocketTTS Italian Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "PocketTTS-GGUF/italian",
"files": [
"PocketTTS-GGUF/italian/pocket-tts-italian-q8_0.gguf"
],
"strip_prefix": "PocketTTS-GGUF/italian"
},
{
"id": "pocket_tts_italian_bf16",
"display_name": "PocketTTS Italian BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "PocketTTS-GGUF/italian",
"files": [
"PocketTTS-GGUF/italian/pocket-tts-italian-bf16.gguf"
],
"strip_prefix": "PocketTTS-GGUF/italian"
},
{
"id": "pocket_tts_portuguese_q8_0",
"display_name": "PocketTTS Portuguese Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "PocketTTS-GGUF/portuguese",
"files": [
"PocketTTS-GGUF/portuguese/pocket-tts-portuguese-q8_0.gguf"
],
"strip_prefix": "PocketTTS-GGUF/portuguese"
},
{
"id": "pocket_tts_portuguese_bf16",
"display_name": "PocketTTS Portuguese BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "PocketTTS-GGUF/portuguese",
"files": [
"PocketTTS-GGUF/portuguese/pocket-tts-portuguese-bf16.gguf"
],
"strip_prefix": "PocketTTS-GGUF/portuguese"
},
{
"id": "pocket_tts_spanish_q8_0",
"display_name": "PocketTTS Spanish Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "PocketTTS-GGUF/spanish",
"files": [
"PocketTTS-GGUF/spanish/pocket-tts-spanish-q8_0.gguf"
],
"strip_prefix": "PocketTTS-GGUF/spanish"
},
{
"id": "pocket_tts_spanish_bf16",
"display_name": "PocketTTS Spanish BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "PocketTTS-GGUF/spanish",
"files": [
"PocketTTS-GGUF/spanish/pocket-tts-spanish-bf16.gguf"
],
"strip_prefix": "PocketTTS-GGUF/spanish"
},
{
"id": "pocket_tts_english_safetensors",
"display_name": "PocketTTS English Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "pocket-tts",
"files": [
"languages/english/embeddings/alba.safetensors",
"languages/english/embeddings/anna.safetensors",
"languages/english/embeddings/azelma.safetensors",
"languages/english/embeddings/bill_boerst.safetensors",
"languages/english/embeddings/caro_davy.safetensors",
"languages/english/embeddings/charles.safetensors",
"languages/english/embeddings/cosette.safetensors",
"languages/english/embeddings/eponine.safetensors",
"languages/english/embeddings/estelle.safetensors",
"languages/english/embeddings/eve.safetensors",
"languages/english/embeddings/fantine.safetensors",
"languages/english/embeddings/george.safetensors",
"languages/english/embeddings/giovanni.safetensors",
"languages/english/embeddings/jane.safetensors",
"languages/english/embeddings/javert.safetensors",
"languages/english/embeddings/jean.safetensors",
"languages/english/embeddings/juergen.safetensors",
"languages/english/embeddings/lola.safetensors",
"languages/english/embeddings/marius.safetensors",
"languages/english/embeddings/mary.safetensors",
"languages/english/embeddings/michael.safetensors",
"languages/english/embeddings/paul.safetensors",
"languages/english/embeddings/peter_yearsley.safetensors",
"languages/english/embeddings/rafael.safetensors",
"languages/english/embeddings/stuart_bell.safetensors",
"languages/english/embeddings/vera.safetensors",
"languages/english/model.safetensors",
"languages/english/tokenizer.model"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "kyutai/pocket-tts"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"tokenizer": "model:tokenizer.model"
},
"optional_files": {
"config": "model:config.yaml"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"language": "languages/english"
},
"files": {
"tokenizer": "language:tokenizer.model"
},
"optional_files": {
"config": "language:config.yaml"
},
"tensors": {
"weights": "language:model.safetensors"
}
}
]
}
+214
View File
@@ -0,0 +1,214 @@
{
"family": "qwen3_asr",
"display_name": "Qwen3-ASR",
"description": "Qwen ASR model family for language identification and speech recognition across 30 languages, 22 Chinese dialects, and multiple English accents, with robustness for noisy, long-form, and singing audio.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"zh",
"en",
"yue",
"ar",
"de",
"fr",
"es",
"pt",
"id",
"it",
"ko",
"ru",
"th",
"vi",
"ja",
"tr",
"hi",
"ms",
"nl",
"sv",
"da",
"fi",
"pl",
"cs",
"fil",
"fa",
"el",
"hu",
"mk",
"ro",
"zh dialects"
],
"capabilities": {
"asr": [
"word_timestamps",
"vad_chunking",
"partial_results"
]
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "qwen3_asr_1_7b_q8_0",
"tags": [
"ASR",
"GGUF",
"Stream"
],
"docs": [
"docs/models/qwen3.md",
"docs/asr.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "qwen3_asr_1_7b_q8_0",
"display_name": "Qwen3-ASR 1.7B Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Qwen3-ASR-1.7B-GGUF",
"files": [
"Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf"
],
"strip_prefix": "Qwen3-ASR-1.7B-GGUF"
},
{
"id": "qwen3_asr_1_7b_f16",
"display_name": "Qwen3-ASR 1.7B F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Qwen3-ASR-1.7B-GGUF",
"files": [
"Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-f16.gguf"
],
"strip_prefix": "Qwen3-ASR-1.7B-GGUF"
},
{
"id": "qwen3_asr_0_6b_q8_0",
"display_name": "Qwen3-ASR 0.6B Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "Qwen3-ASR-0.6B-GGUF",
"files": [
"Qwen3-ASR-0.6B-GGUF/qwen3-asr-0.6b-q8_0.gguf"
],
"strip_prefix": "Qwen3-ASR-0.6B-GGUF"
},
{
"id": "qwen3_asr_0_6b_f16",
"display_name": "Qwen3-ASR 0.6B F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Qwen3-ASR-0.6B-GGUF",
"files": [
"Qwen3-ASR-0.6B-GGUF/qwen3-asr-0.6b-f16.gguf"
],
"strip_prefix": "Qwen3-ASR-0.6B-GGUF"
},
{
"id": "qwen3_asr_1_7b_safetensors",
"display_name": "Qwen3-ASR 1.7B HF Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "Qwen3-ASR-1.7B-hf",
"files": [
"config.json",
"generation_config.json",
"model.safetensors",
"processor_config.json",
"tokenizer_config.json",
"tokenizer.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "Qwen/Qwen3-ASR-1.7B-hf"
}
},
{
"id": "qwen3_asr_0_6b_safetensors",
"display_name": "Qwen3-ASR 0.6B Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "Qwen3-ASR-0.6B",
"files": [
"config.json",
"generation_config.json",
"model.safetensors",
"preprocessor_config.json",
"tokenizer_config.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "Qwen/Qwen3-ASR-0.6B"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer_config": "model:tokenizer_config.json"
},
"optional_files": {
"preprocessor_config": "model:preprocessor_config.json",
"processor_config": "model:processor_config.json",
"chat_template": "model:chat_template.json",
"chat_template_jinja": "model:chat_template.jinja",
"vocab": "model:vocab.json",
"merges": "model:merges.txt",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer_config": "model:tokenizer_config.json"
},
"optional_files": {
"preprocessor_config": "model:preprocessor_config.json",
"processor_config": "model:processor_config.json",
"chat_template": "model:chat_template.json",
"chat_template_jinja": "model:chat_template.jinja",
"vocab": "model:vocab.json",
"merges": "model:merges.txt",
"tokenizer_json": "model:tokenizer.json"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
+225
View File
@@ -0,0 +1,225 @@
{
"family": "qwen3_tts",
"display_name": "Qwen3-TTS",
"description": "Qwen TTS family for controllable 10-language speech synthesis, including 3-second voice cloning, CustomVoice instruction control over preset timbres, and VoiceDesign from natural-language descriptions.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone",
"design"
],
"modes": [
"offline"
],
"languages": [
"zh",
"en",
"ja",
"ko",
"de",
"fr",
"ru",
"pt",
"es",
"it"
],
"capabilities": {
"clone": [
"speaker_reference"
],
"design": [
"voice_design"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "qwen3_tts_1_7b_base_q8_0",
"tags": [
"TTS",
"Clone",
"Design",
"GGUF"
],
"docs": [
"docs/models/qwen3.md",
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "qwen3_tts_1_7b_base_q8_0",
"display_name": "Qwen3 TTS 12Hz 1.7B Base Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-q8_0_v2.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
},
{
"id": "qwen3_tts_1_7b_base_bf16",
"display_name": "Qwen3 TTS 12Hz 1.7B Base BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-bf16.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
},
{
"id": "qwen3_tts_1_7b_base_orig",
"display_name": "Qwen3 TTS 12Hz 1.7B Base Original-Dtype GGUF",
"format": "gguf",
"precision": "orig",
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-orig.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
},
{
"id": "qwen3_tts_1_7b_customvoice_q8_0",
"display_name": "Qwen3 TTS 12Hz 1.7B CustomVoice Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q8_0.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF"
},
{
"id": "qwen3_tts_1_7b_customvoice_bf16",
"display_name": "Qwen3 TTS 12Hz 1.7B CustomVoice BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-bf16.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF"
},
{
"id": "qwen3_tts_1_7b_voicedesign_q8_0",
"display_name": "Qwen3 TTS 12Hz 1.7B VoiceDesign Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF/qwen3-tts-12hz-1.7b-voicedesign-q8_0.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF"
},
{
"id": "qwen3_tts_1_7b_voicedesign_bf16",
"display_name": "Qwen3 TTS 12Hz 1.7B VoiceDesign BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF",
"files": [
"Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF/qwen3-tts-12hz-1.7b-voicedesign-bf16.gguf"
],
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF"
},
{
"id": "qwen3_tts_1_7b_base_safetensors",
"display_name": "Qwen3 TTS 12Hz 1.7B Base Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base",
"files": [
"config.json",
"generation_config.json",
"model.safetensors",
"speech_tokenizer/config.json",
"speech_tokenizer/model.safetensors",
"tokenizer_config.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
}
},
{
"id": "qwen3_tts_0_6b_base_safetensors",
"display_name": "Qwen3 TTS 12Hz 0.6B Base Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "Qwen3-TTS-12Hz-0.6B-Base",
"files": [
"config.json",
"generation_config.json",
"model.safetensors",
"speech_tokenizer/config.json",
"speech_tokenizer/model.safetensors",
"tokenizer_config.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "Qwen/Qwen3-TTS-12Hz-0.6B-Base"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt",
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"speech_tokenizer_weights": {
"source": "weights:",
"prefix": "speech_tokenizer_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt",
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
},
"tensors": {
"model_weights": "model:model.safetensors",
"speech_tokenizer_weights": "model:speech_tokenizer/model.safetensors"
}
}
]
}
+337
View File
@@ -0,0 +1,337 @@
{
"family": "seed_vc",
"schema_version": 1,
"display_name": "Seed-VC",
"description": "Zero-shot voice conversion and singing voice conversion model for transferring timbre and style from reference audio, with low-latency realtime conversion and optional lightweight fine-tuning.",
"category": "voice_conversion",
"status": "supported",
"tasks": [
"vc",
"svc"
],
"modes": [
"offline"
],
"languages": [
"language_agnostic"
],
"capabilities": {
"vc": [
"speaker_reference"
],
"svc": [
"speaker_reference",
"singing"
]
},
"dependencies": [],
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "seed_vc_mlx_q8_0",
"tags": [
"VC",
"GGUF"
],
"docs": [
"docs/models/seed_vc.md",
"docs/audio_tools.md",
"docs/gguf.md"
]
},
"options": {
"request": [
{
"name": "route",
"type": "enum",
"description": "Select the Seed-VC conversion route. Defaults to v2_vc for VC and v1_svc for SVC.",
"values": [
"v2_vc",
"v1_svc",
"v1_whisper_bigvgan_vc",
"v1_xlsr_hift_vc"
],
"required": false
},
{
"name": "length_adjust",
"type": "float",
"description": "Output duration multiplier; must be positive, default 1.0.",
"required": false,
"min": 0.0,
"default": 1.0
},
{
"name": "num_inference_steps",
"type": "int",
"description": "Diffusion steps; default 30.",
"required": false,
"min": 1,
"default": 30
},
{
"name": "inference_guidance_scale",
"type": "float",
"description": "V1 classifier-free guidance scale; default 0.7.",
"required": false,
"min": 0.0,
"default": 0.7
},
{
"name": "intelligibility_guidance_scale",
"type": "float",
"description": "V2 classifier-free guidance scale for source-content intelligibility; default 0.7.",
"required": false,
"min": 0.0,
"default": 0.7
},
{
"name": "similarity_guidance_scale",
"type": "float",
"description": "V2 classifier-free guidance scale for target-speaker similarity; default 0.7.",
"required": false,
"min": 0.0,
"default": 0.7
},
{
"name": "voice_anonymization",
"type": "bool",
"description": "Use randomized average-voice conditioning instead of target-speaker conditioning for V2 anonymization; default false.",
"required": false,
"default": false
},
{
"name": "seed",
"type": "int",
"description": "Seed for V1/V2 diffusion noise and HiFT stochastic source excitation; omitted requests choose a random seed.",
"required": false,
"min": 0
},
{
"name": "noise_path",
"type": "path",
"description": "Optional raw f32 noise file for deterministic V1/V2 diffusion noise and XLSR/HiFT source excitation.",
"required": false
},
{
"name": "f0_condition",
"type": "bool",
"description": "Enable V1 F0 conditioning for singing voice conversion; default false.",
"required": false,
"default": false
},
{
"name": "auto_f0_adjust",
"type": "bool",
"description": "Automatically adjust V1 source pitch toward the target pitch level; default false.",
"required": false,
"default": false
},
{
"name": "semitone_shift",
"type": "int",
"description": "V1 pitch shift in semitones for singing voice conversion; default 0.",
"required": false,
"default": 0
}
],
"session": [
{
"name": "weight_type",
"type": "enum",
"description": "Shared Seed-VC component weight storage type; default native, except RMVPE uses f32 unless overridden.",
"preset": "weight_type_full",
"required": false,
"default": "native"
}
],
"load": []
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "seed_vc_mlx_q8_0",
"display_name": "SeedVC-MLX Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "SeedVC-MLX-GGUF",
"files": [
"SeedVC-MLX-GGUF/seed-vc-mlx-q8_0.gguf"
],
"strip_prefix": "SeedVC-MLX-GGUF"
},
{
"id": "seed_vc_mlx_f16",
"display_name": "SeedVC-MLX F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "SeedVC-MLX-GGUF",
"files": [
"SeedVC-MLX-GGUF/seed-vc-mlx-f16.gguf"
],
"strip_prefix": "SeedVC-MLX-GGUF"
},
{
"id": "seed_vc_mlx_orig",
"display_name": "SeedVC-MLX Original-Dtype GGUF",
"format": "gguf",
"precision": "orig",
"target_directory": "SeedVC-MLX-GGUF",
"files": [
"SeedVC-MLX-GGUF/seed-vc-mlx-orig.gguf"
],
"strip_prefix": "SeedVC-MLX-GGUF"
},
{
"id": "seed_vc_mlx_safetensors",
"display_name": "SeedVC-MLX Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "SeedVC-MLX",
"files": [
"seed_vc_manifest.json",
"v2/ar.safetensors",
"v2/cfm.safetensors"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "mlx-community/SeedVC-MLX"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"manifest": "model:seed_vc_manifest.json",
"v2_wrapper_config": "model:v2/vc_wrapper.json",
"astral_bsq32_config": "model:astral/bsq32.json",
"astral_bsq2048_config": "model:astral/bsq2048.json",
"v1_svc_config": "model:v1/svc.json",
"v1_whisper_bigvgan_config": "model:v1/whisper_bigvgan.json",
"v1_xlsr_hift_config": "model:v1/xlsr_hift.json",
"hift_config": "model:hift/config.json",
"bigvgan_22k_config": "model:bigvgan/v2_22khz_80band_256x/config.json",
"bigvgan_44k_config": "model:bigvgan/v2_44khz_128band_512x/config.json",
"whisper_small_config": "model:whisper-small/config.json",
"hubert_large_config": "model:hubert-large-ll60k/config.json",
"wav2vec2_xlsr_config": "model:wav2vec2-xls-r-300m/config.json"
},
"tensors": {
"v2_ar_weights": {
"source": "weights:",
"prefix": "v2_ar_weights"
},
"v2_cfm_weights": {
"source": "weights:",
"prefix": "v2_cfm_weights"
},
"v1_svc_weights": {
"source": "weights:",
"prefix": "v1_svc_weights"
},
"v1_whisper_bigvgan_weights": {
"source": "weights:",
"prefix": "v1_whisper_bigvgan_weights"
},
"v1_xlsr_hift_weights": {
"source": "weights:",
"prefix": "v1_xlsr_hift_weights"
},
"astral_bsq32_weights": {
"source": "weights:",
"prefix": "astral_bsq32_weights"
},
"astral_bsq2048_weights": {
"source": "weights:",
"prefix": "astral_bsq2048_weights"
},
"campplus_weights": {
"source": "weights:",
"prefix": "campplus_weights"
},
"rmvpe_weights": {
"source": "weights:",
"prefix": "rmvpe_weights"
},
"hift_weights": {
"source": "weights:",
"prefix": "hift_weights"
},
"bigvgan_22k_weights": {
"source": "weights:",
"prefix": "bigvgan_22k_weights"
},
"bigvgan_44k_weights": {
"source": "weights:",
"prefix": "bigvgan_44k_weights"
},
"whisper_small_weights": {
"source": "weights:",
"prefix": "whisper_small_weights"
},
"hubert_large_weights": {
"source": "weights:",
"prefix": "hubert_large_weights"
},
"wav2vec2_xlsr_weights": {
"source": "weights:",
"prefix": "wav2vec2_xlsr_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"manifest": "model:seed_vc_manifest.json",
"v2_wrapper_config": "model:v2/vc_wrapper.json",
"astral_bsq32_config": "model:astral/bsq32.json",
"astral_bsq2048_config": "model:astral/bsq2048.json",
"v1_svc_config": "model:v1/svc.json",
"v1_whisper_bigvgan_config": "model:v1/whisper_bigvgan.json",
"v1_xlsr_hift_config": "model:v1/xlsr_hift.json",
"hift_config": "model:hift/config.json",
"bigvgan_22k_config": "model:bigvgan/v2_22khz_80band_256x/config.json",
"bigvgan_44k_config": "model:bigvgan/v2_44khz_128band_512x/config.json",
"whisper_small_config": "model:whisper-small/config.json",
"hubert_large_config": "model:hubert-large-ll60k/config.json",
"wav2vec2_xlsr_config": "model:wav2vec2-xls-r-300m/config.json"
},
"tensors": {
"v2_ar_weights": "model:v2/ar.safetensors",
"v2_cfm_weights": "model:v2/cfm.safetensors",
"v1_svc_weights": "model:v1/svc.safetensors",
"v1_whisper_bigvgan_weights": "model:v1/whisper_bigvgan.safetensors",
"v1_xlsr_hift_weights": "model:v1/xlsr_hift.safetensors",
"astral_bsq32_weights": "model:astral/bsq32.safetensors",
"astral_bsq2048_weights": "model:astral/bsq2048.safetensors",
"campplus_weights": "model:campplus/model.safetensors",
"rmvpe_weights": "model:rmvpe/model.safetensors",
"hift_weights": "model:hift/model.safetensors",
"bigvgan_22k_weights": "model:bigvgan/v2_22khz_80band_256x/model.safetensors",
"bigvgan_44k_weights": "model:bigvgan/v2_44khz_128band_512x/model.safetensors",
"whisper_small_weights": "model:whisper-small/model.safetensors",
"hubert_large_weights": "model:hubert-large-ll60k/model.safetensors",
"wav2vec2_xlsr_weights": "model:wav2vec2-xls-r-300m/model.safetensors"
}
}
]
}
+183
View File
@@ -0,0 +1,183 @@
{
"family": "supertonic",
"display_name": "Supertonic 3",
"description": "Supertone on-device TTS model designed for fast local speech synthesis across 31 languages, with preset voices and compact deployment for browser, mobile, and desktop applications.",
"category": "tts",
"status": "supported",
"tasks": [
"tts"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"en",
"ko",
"ja",
"ar",
"bg",
"cs",
"da",
"de",
"el",
"es",
"et",
"fi",
"fr",
"hi",
"hr",
"hu",
"id",
"it",
"lt",
"lv",
"nl",
"pl",
"pt",
"ro",
"ru",
"sk",
"sl",
"sv",
"tr",
"uk",
"vi"
],
"capabilities": {
"tts": [
"built_in_voices",
"long_form"
]
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "supertonic_3_orig",
"tags": [
"TTS",
"GGUF",
"Stream"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "supertonic_3_q8_0",
"display_name": "Supertonic 3 Q8_0 GGUF",
"format": "gguf",
"precision": "q8_0",
"target_directory": "Supertonic-3-GGUF",
"files": [
"Supertonic-3-GGUF/supertonic-3-q8_0.gguf"
],
"strip_prefix": "Supertonic-3-GGUF"
},
{
"id": "supertonic_3_f16",
"display_name": "Supertonic 3 F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Supertonic-3-GGUF",
"files": [
"Supertonic-3-GGUF/supertonic-3-f16.gguf"
],
"strip_prefix": "Supertonic-3-GGUF"
},
{
"id": "supertonic_3_orig",
"display_name": "Supertonic 3 Original-Dtype GGUF",
"default": true,
"format": "gguf",
"precision": "orig",
"target_directory": "Supertonic-3-GGUF",
"files": [
"Supertonic-3-GGUF/supertonic-3-orig.gguf"
],
"strip_prefix": "Supertonic-3-GGUF"
},
{
"id": "supertonic_3_safetensors",
"display_name": "Supertonic 3 Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "supertonic-3",
"files": [
"config/tts.json",
"config/unicode_indexer.json",
"ggml/supertonic.safetensors"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "mlx-community/supertonic-3-mlx"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"tts_config": "model:config/tts.json",
"unicode_indexer": "model:config/unicode_indexer.json",
"voice_style_F1": "model:voice_styles/F1.json",
"voice_style_F2": "model:voice_styles/F2.json",
"voice_style_F3": "model:voice_styles/F3.json",
"voice_style_F4": "model:voice_styles/F4.json",
"voice_style_F5": "model:voice_styles/F5.json",
"voice_style_M1": "model:voice_styles/M1.json",
"voice_style_M2": "model:voice_styles/M2.json",
"voice_style_M3": "model:voice_styles/M3.json",
"voice_style_M4": "model:voice_styles/M4.json",
"voice_style_M5": "model:voice_styles/M5.json"
},
"tensors": {
"weights": {
"source": "weights:",
"prefix": "weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"tts_config": "model:config/tts.json",
"unicode_indexer": "model:config/unicode_indexer.json",
"voice_style_F1": "model:voice_styles/F1.json",
"voice_style_F2": "model:voice_styles/F2.json",
"voice_style_F3": "model:voice_styles/F3.json",
"voice_style_F4": "model:voice_styles/F4.json",
"voice_style_F5": "model:voice_styles/F5.json",
"voice_style_M1": "model:voice_styles/M1.json",
"voice_style_M2": "model:voice_styles/M2.json",
"voice_style_M3": "model:voice_styles/M3.json",
"voice_style_M4": "model:voice_styles/M4.json",
"voice_style_M5": "model:voice_styles/M5.json"
},
"tensors": {
"weights": "model:ggml/supertonic.safetensors"
}
}
]
}
+209
View File
@@ -0,0 +1,209 @@
{
"family": "vevo2",
"display_name": "Vevo2",
"description": "Unified controllable framework for English and Chinese speech and singing voice generation, voice conversion, and editing, with tokenizers that disentangle content, prosody, melody, style, and timbre.",
"category": "voice_conversion",
"status": "supported",
"tasks": [
"tts",
"music",
"vc",
"edit",
"svc",
"s2s"
],
"modes": [
"offline"
],
"languages": [
"en",
"zh"
],
"capabilities": {
"music": [
"lyrics"
],
"vc": [
"speaker_reference"
],
"svc": [
"speaker_reference",
"singing"
],
"s2s": [
"speaker_reference"
],
"edit": [
"prompt_editing"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "vevo2_q8_0",
"tags": [
"TTS",
"Music",
"VC",
"Edit",
"GGUF"
],
"docs": [
"docs/models/vevo2.md",
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "vevo2_q8_0",
"display_name": "Vevo2 Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Vevo2-GGUF",
"files": [
"Vevo2-GGUF/vevo2-q8_0.gguf"
],
"strip_prefix": "Vevo2-GGUF"
},
{
"id": "vevo2_f16",
"display_name": "Vevo2 F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "Vevo2-GGUF",
"files": [
"Vevo2-GGUF/vevo2-f16.gguf"
],
"strip_prefix": "Vevo2-GGUF"
},
{
"id": "vevo2_orig",
"display_name": "Vevo2 Original-Dtype GGUF",
"format": "gguf",
"precision": "orig",
"target_directory": "Vevo2-GGUF",
"files": [
"Vevo2-GGUF/vevo2-orig.gguf"
],
"strip_prefix": "Vevo2-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"ar_config": "model:contentstyle_modeling/posttrained/config.json",
"ar_amphion_config": "model:contentstyle_modeling/posttrained/amphion_config.json",
"ar_generation_config": "model:contentstyle_modeling/posttrained/generation_config.json",
"ar_tokenizer_config": "model:contentstyle_modeling/posttrained/tokenizer_config.json",
"ar_tokenizer_json": "model:contentstyle_modeling/posttrained/tokenizer.json",
"ar_vocab": "model:contentstyle_modeling/posttrained/vocab.json",
"ar_merges": "model:contentstyle_modeling/posttrained/merges.txt",
"ar_added_tokens": "model:contentstyle_modeling/posttrained/added_tokens.json",
"ar_special_tokens": "model:contentstyle_modeling/posttrained/special_tokens_map.json",
"fm_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/config.json",
"fm_text_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/config.json",
"vocoder_config": "model:vocoder/config.json",
"whisper_config": "model:whisper-medium/config.json"
},
"tensors": {
"content_style_tokenizer_weights": {
"source": "weights:",
"prefix": "content_style_tokenizer_weights"
},
"prosody_tokenizer_weights": {
"source": "weights:",
"prefix": "prosody_tokenizer_weights"
},
"ar_weights": {
"source": "weights:",
"prefix": "ar_weights"
},
"fm_weights": {
"source": "weights:",
"prefix": "fm_weights"
},
"fm_whisper_stats": {
"source": "weights:",
"prefix": "fm_whisper_stats"
},
"fm_text_weights": {
"source": "weights:",
"prefix": "fm_text_weights"
},
"fm_text_whisper_stats": {
"source": "weights:",
"prefix": "fm_text_whisper_stats"
},
"vocoder_weights_0": {
"source": "weights:",
"prefix": "vocoder_weights_0"
},
"vocoder_weights_1": {
"source": "weights:",
"prefix": "vocoder_weights_1"
},
"vocoder_weights_2": {
"source": "weights:",
"prefix": "vocoder_weights_2"
},
"whisper_weights": {
"source": "weights:",
"prefix": "whisper_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": ".",
"whisper": "../whisper-medium"
},
"files": {
"ar_config": "model:contentstyle_modeling/posttrained/config.json",
"ar_amphion_config": "model:contentstyle_modeling/posttrained/amphion_config.json",
"ar_generation_config": "model:contentstyle_modeling/posttrained/generation_config.json",
"ar_tokenizer_config": "model:contentstyle_modeling/posttrained/tokenizer_config.json",
"ar_tokenizer_json": "model:contentstyle_modeling/posttrained/tokenizer.json",
"ar_vocab": "model:contentstyle_modeling/posttrained/vocab.json",
"ar_merges": "model:contentstyle_modeling/posttrained/merges.txt",
"ar_added_tokens": "model:contentstyle_modeling/posttrained/added_tokens.json",
"ar_special_tokens": "model:contentstyle_modeling/posttrained/special_tokens_map.json",
"fm_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/config.json",
"fm_text_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/config.json",
"vocoder_config": "model:vocoder/config.json",
"whisper_config": "whisper:config.json"
},
"tensors": {
"content_style_tokenizer_weights": "model:tokenizer/contentstyle_fvq16384_12.5hz/model.safetensors",
"prosody_tokenizer_weights": "model:tokenizer/prosody_fvq512_6.25hz/model.safetensors",
"ar_weights": "model:contentstyle_modeling/posttrained/model.safetensors",
"fm_weights": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/model.safetensors",
"fm_whisper_stats": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/whisper_stats.safetensors",
"fm_text_weights": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/model.safetensors",
"fm_text_whisper_stats": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/whisper_stats.safetensors",
"vocoder_weights_0": "model:vocoder/model.safetensors",
"vocoder_weights_1": "model:vocoder/model_1.safetensors",
"vocoder_weights_2": "model:vocoder/model_2.safetensors",
"whisper_weights": "whisper:model.safetensors"
}
}
]
}
+109
View File
@@ -0,0 +1,109 @@
{
"family": "vibevoice",
"display_name": "VibeVoice",
"description": "Microsoft long-form multi-speaker TTS model for expressive conversational audio such as podcasts, supporting up to 90 minutes of speech with as many as four speakers.",
"category": "tts",
"status": "supported",
"tasks": [
"tts"
],
"modes": [
"offline"
],
"languages": [
"en",
"zh"
],
"capabilities": {
"tts": [
"multi_speaker",
"long_form"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "vibevoice_1_5b_q8_0",
"tags": [
"TTS",
"GGUF"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "vibevoice_1_5b_q8_0",
"display_name": "VibeVoice 1.5B Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "VibeVoice-1.5B-GGUF",
"files": [
"VibeVoice-1.5B-GGUF/vibevoice-1.5b-q8_0.gguf"
],
"strip_prefix": "VibeVoice-1.5B-GGUF"
},
{
"id": "vibevoice_1_5b_bf16",
"display_name": "VibeVoice 1.5B BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "VibeVoice-1.5B-GGUF",
"files": [
"VibeVoice-1.5B-GGUF/vibevoice-1.5b-bf16.gguf"
],
"strip_prefix": "VibeVoice-1.5B-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"preprocessor_config": "model:preprocessor_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_vocab": "model:vocab.json",
"tokenizer_merges": "model:merges.txt"
},
"tensors": {
"model_weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"preprocessor_config": "model:preprocessor_config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_vocab": "model:vocab.json",
"tokenizer_merges": "model:merges.txt"
},
"tensors": {
"model_weights": "model:model.safetensors.index.json"
}
}
]
}
@@ -0,0 +1,114 @@
{
"family": "vibevoice_asr",
"display_name": "VibeVoice ASR",
"description": "Microsoft long-form speech-to-text model that processes up to 60 minutes of audio in one pass and produces structured transcripts with speakers, timestamps, content, hotwords, and 50+ language support.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline"
],
"languages": [
"auto",
"51 languages"
],
"capabilities": {
"asr": [
"segments",
"speaker_turns",
"vad_chunking"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "vibevoice_asr_q8_0",
"tags": [
"ASR",
"GGUF"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "vibevoice_asr_q8_0",
"display_name": "VibeVoice ASR Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "VibeVoice-ASR-GGUF",
"files": [
"VibeVoice-ASR-GGUF/vibevoice-asr-q8_0.gguf"
],
"strip_prefix": "VibeVoice-ASR-GGUF"
},
{
"id": "vibevoice_asr_f16",
"display_name": "VibeVoice ASR F16 GGUF",
"format": "gguf",
"precision": "f16",
"target_directory": "VibeVoice-ASR-GGUF",
"files": [
"VibeVoice-ASR-GGUF/vibevoice-asr-f16.gguf"
],
"strip_prefix": "VibeVoice-ASR-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_vocab": "model:vocab.json",
"tokenizer_merges": "model:merges.txt"
},
"optional_files": {
"preprocessor_config": "model:preprocessor_config.json"
},
"tensors": {
"model_weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json",
"tokenizer_vocab": "model:vocab.json",
"tokenizer_merges": "model:merges.txt"
},
"optional_files": {
"preprocessor_config": "model:preprocessor_config.json"
},
"tensors": {
"model_weights": "model:model.safetensors.index.json"
}
}
]
}
@@ -0,0 +1,112 @@
{
"family": "vietneu_tts",
"display_name": "VieNeu-TTS v3 Turbo",
"description": "On-device Vietnamese TTS model with instant voice cloning from 3-5 seconds of reference audio, English-Vietnamese code-switching, streaming playback, batched generation, and conversation mode.",
"category": "tts",
"status": "community",
"tasks": [
"tts",
"clone"
],
"modes": [
"offline"
],
"languages": [
"vi",
"en"
],
"capabilities": {
"clone": [
"speaker_reference"
]
},
"runtime": {
"tags": [
"gguf"
]
},
"ui": {
"recommended_package": "vietneu_tts_v3_turbo_q8_0",
"tags": [
"TTS",
"Clone",
"GGUF"
],
"docs": [
"docs/community_models/vietneu_tts.md",
"docs/tts.md",
"docs/gguf.md"
]
},
"packages": [
{
"id": "vietneu_tts_v3_turbo_q8_0",
"display_name": "VieNeu-TTS v3 Turbo GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "VieNeu-TTS-v3-Turbo-GGUF",
"files": [
"model.gguf"
],
"strip_prefix": ".",
"download": {
"kind": "huggingface_snapshot",
"repo": "phuocnguyen90/VieNeu-TTS-v3-Turbo-GGUF"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
},
"optional_files": {
"generation_config": "model:generation_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt",
"tokenizer_json": "model:tokenizer.json",
"special_tokens_map": "model:special_tokens_map.json"
},
"tensors": {
"model_weights": {
"source": "weights:",
"prefix": "model_weights"
},
"speech_tokenizer_weights": {
"source": "weights:",
"prefix": "speech_tokenizer_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
},
"optional_files": {
"generation_config": "model:generation_config.json",
"vocab": "model:vocab.json",
"merges": "model:merges.txt",
"tokenizer_json": "model:tokenizer.json",
"special_tokens_map": "model:special_tokens_map.json"
},
"tensors": {
"model_weights": "model:model.safetensors",
"speech_tokenizer_weights": "model:speech_tokenizer/model.safetensors"
}
}
]
}
+179
View File
@@ -0,0 +1,179 @@
{
"family": "voxcpm2",
"display_name": "VoxCPM2",
"description": "OpenBMB tokenizer-free TTS model supporting 30 languages and 9 Chinese dialects, with 48 kHz output, natural-language voice design, controllable short-reference voice cloning, and expressive style guidance.",
"category": "tts",
"status": "supported",
"tasks": [
"tts",
"clone",
"design"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"ar",
"my",
"zh",
"zh dialects",
"da",
"nl",
"en",
"fi",
"fr",
"de",
"el",
"he",
"hi",
"id",
"it",
"ja",
"km",
"ko",
"lo",
"ms",
"no",
"pl",
"pt",
"ru",
"es",
"sw",
"sv",
"tl",
"th",
"tr",
"vi"
],
"capabilities": {
"clone": [
"speaker_reference"
],
"design": [
"voice_design"
]
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "voxcpm2_q8_0",
"tags": [
"TTS",
"Clone",
"Design",
"GGUF",
"Stream"
],
"docs": [
"docs/tts.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "voxcpm2_q8_0",
"display_name": "VoxCPM2 Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "VoxCPM2-GGUF",
"files": [
"VoxCPM2-GGUF/voxcpm2-q8_0.gguf"
],
"strip_prefix": "VoxCPM2-GGUF"
},
{
"id": "voxcpm2_bf16",
"display_name": "VoxCPM2 BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "VoxCPM2-GGUF",
"files": [
"VoxCPM2-GGUF/voxcpm2-bf16.gguf"
],
"strip_prefix": "VoxCPM2-GGUF"
},
{
"id": "voxcpm2_orig",
"display_name": "VoxCPM2 Original-Dtype GGUF",
"format": "gguf",
"precision": "orig",
"target_directory": "VoxCPM2-GGUF",
"files": [
"VoxCPM2-GGUF/voxcpm2-orig.gguf"
],
"strip_prefix": "VoxCPM2-GGUF"
},
{
"id": "voxcpm2_safetensors",
"display_name": "VoxCPM2 Safetensors",
"format": "safetensors",
"precision": "native",
"target_directory": "VoxCPM2",
"files": [
"config.json",
"model.safetensors",
"tokenizer.json",
"tokenizer_config.json"
],
"download": {
"kind": "huggingface_snapshot",
"repo": "OpenBMB/VoxCPM2"
}
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json",
"special_tokens_map": "model:special_tokens_map.json"
},
"tensors": {
"weights": {
"source": "weights:",
"prefix": "weights"
},
"audiovae_weights": {
"source": "weights:",
"prefix": "audiovae_weights"
}
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"tokenizer_config": "model:tokenizer_config.json",
"tokenizer_json": "model:tokenizer.json",
"special_tokens_map": "model:special_tokens_map.json"
},
"tensors": {
"weights": "model:model.safetensors",
"audiovae_weights": "model:audiovae.safetensors"
}
}
]
}
@@ -0,0 +1,137 @@
{
"family": "voxtral_realtime",
"display_name": "Voxtral Mini 4B Realtime",
"description": "Mistral 13-language realtime ASR model with a natively streaming causal audio encoder, configurable low-latency transcription delay, and accuracy competitive with offline open-source systems.",
"category": "asr",
"status": "supported",
"tasks": [
"asr"
],
"modes": [
"offline",
"streaming"
],
"languages": [
"en",
"zh",
"hi",
"es",
"ar",
"fr",
"pt",
"ru",
"de",
"ja",
"ko",
"it",
"nl"
],
"capabilities": {
"asr": [
"partial_results"
]
},
"runtime": {
"tags": [
"gguf",
"stream"
]
},
"ui": {
"recommended_package": "voxtral_realtime_q8_0",
"tags": [
"ASR",
"GGUF",
"Stream"
],
"docs": [
"docs/asr.md",
"docs/gguf.md"
]
},
"package_defaults": {
"download": {
"kind": "huggingface_snapshot",
"repo": "audio-cpp/audio.cpp-gguf",
"revision": "main",
"gated": false
}
},
"packages": [
{
"id": "voxtral_realtime_q8_0",
"display_name": "Voxtral Mini 4B Realtime Q8_0 GGUF",
"default": true,
"format": "gguf",
"precision": "q8_0",
"target_directory": "Voxtral-Mini-4B-Realtime-2602-GGUF",
"files": [
"Voxtral-Mini-4B-Realtime-2602-GGUF/voxtral-mini-4b-realtime-2602-q8_0.gguf"
],
"strip_prefix": "Voxtral-Mini-4B-Realtime-2602-GGUF"
},
{
"id": "voxtral_realtime_q4_k",
"display_name": "Voxtral Mini 4B Realtime Q4_K GGUF",
"format": "gguf",
"precision": "q4_k",
"target_directory": "Voxtral-Mini-4B-Realtime-2602-GGUF",
"files": [
"Voxtral-Mini-4B-Realtime-2602-GGUF/voxtral-mini-4b-realtime-2602-q4_k.gguf"
],
"strip_prefix": "Voxtral-Mini-4B-Realtime-2602-GGUF"
},
{
"id": "voxtral_realtime_bf16",
"display_name": "Voxtral Mini 4B Realtime BF16 GGUF",
"format": "gguf",
"precision": "bf16",
"target_directory": "Voxtral-Mini-4B-Realtime-2602-GGUF",
"files": [
"Voxtral-Mini-4B-Realtime-2602-GGUF/voxtral-mini-4b-realtime-2602-bf16.gguf"
],
"strip_prefix": "Voxtral-Mini-4B-Realtime-2602-GGUF"
}
],
"sources": [
{
"format": "gguf",
"roots": {
"model": ".",
"weights": "$gguf"
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"processor_config": "model:processor_config.json",
"tekken": "model:tekken.json"
},
"optional_files": {
"params": "model:params.json",
"readme": "model:README.md"
},
"tensors": {
"weights": "weights:"
}
},
{
"format": "safetensors",
"roots": {
"model": "."
},
"files": {
"config": "model:config.json",
"generation_config": "model:generation_config.json",
"processor_config": "model:processor_config.json",
"tekken": "model:tekken.json"
},
"optional_files": {
"params": "model:params.json",
"readme": "model:README.md"
},
"tensors": {
"weights": "model:model.safetensors"
}
}
]
}
+446
View File
@@ -0,0 +1,446 @@
"""Owned audio.cpp server process launcher."""
from __future__ import annotations
import json
import os
import shutil
import socket
import subprocess
import tempfile
import threading
import time
from pathlib import Path
from typing import Any, Dict, Mapping, Optional, Sequence
from .client import AudioCppClient, AudioCppClientError
from .windows_job import WindowsKillOnCloseJob
class AudioCppProcessError(RuntimeError):
"""Base error for managed audio.cpp process failures."""
class AudioCppProcessStartupError(AudioCppProcessError):
"""The owned audio.cpp server failed before becoming ready."""
def find_free_loopback_port() -> int:
"""Ask the OS for a currently unused IPv4 loopback port."""
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
sock.bind(("127.0.0.1", 0))
return int(sock.getsockname()[1])
def normalize_audio_cpp_task(task: Any) -> str:
normalized = str(task or "tts").strip().lower().replace("-", "_").replace(" ", "_")
aliases = {
"clone": "clon",
"cloning": "clon",
"voice_clone": "clon",
"voice_cloning": "clon",
"design": "vdes",
"voice_design": "vdes",
"voice_designer": "vdes",
}
return aliases.get(normalized, normalized)
def _first(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
for key in keys:
if key in config and config[key] is not None:
return config[key]
return default
def _resolve_existing_path(value: Any, label: str, *, executable: bool = False) -> Path:
raw = os.path.expandvars(os.path.expanduser(str(value or "").strip()))
if not raw:
raise AudioCppProcessStartupError(f"Missing audio.cpp {label}")
candidate = Path(raw)
if executable and not candidate.exists():
located = shutil.which(raw)
if located:
candidate = Path(located)
candidate = candidate.resolve()
if not candidate.exists():
raise AudioCppProcessStartupError(f"audio.cpp {label} does not exist: {candidate}")
if executable and not candidate.is_file():
raise AudioCppProcessStartupError(f"audio.cpp {label} is not a file: {candidate}")
return candidate
def _safe_model_id(value: Any, family: str) -> str:
raw = str(value or family or "audio-cpp-model").strip()
cleaned = "".join(char if char.isalnum() or char in "._-" else "-" for char in raw)
cleaned = cleaned.strip("-.")
return cleaned or "audio-cpp-model"
def _option_dict(config: Mapping[str, Any], key: str) -> Dict[str, Any]:
value = config.get(key, {})
if value is None:
return {}
if not isinstance(value, Mapping):
raise AudioCppProcessStartupError(f"audio.cpp {key} must be a JSON object")
return dict(value)
class AudioCppServerProcess:
"""One suite-owned, single-model ``audiocpp_server`` process."""
def __init__(self, config: Mapping[str, Any]) -> None:
self.config = dict(config)
self.binary_path = _resolve_existing_path(
_first(
self.config,
"binary_path",
"server_binary",
"executable_path",
"audio_cpp_binary",
),
"server executable",
executable=True,
)
self.model_path = _resolve_existing_path(
_first(self.config, "model_path", "package_path", "gguf_path"),
"model path",
)
self.family = str(_first(self.config, "family", "model_family", default="")).strip()
if not self.family:
raise AudioCppProcessStartupError("Missing audio.cpp model family")
self.task = normalize_audio_cpp_task(_first(self.config, "task", default="tts"))
self.model_id = _safe_model_id(
_first(self.config, "model_id", "server_model_id", "package_id"),
self.family,
)
self.backend = str(_first(self.config, "backend", default="cuda")).strip().lower()
if self.backend == "auto":
self.backend = "cuda"
if self.backend not in {"cuda", "hip", "cpu", "vulkan", "metal"}:
raise AudioCppProcessStartupError(f"Unsupported audio.cpp backend: {self.backend}")
self.device = self._resolve_device_index(
_first(self.config, "device_index", "device", default=0)
)
# Match audio.cpp's CLI default instead of the server's conservative
# one-thread example configuration.
self.threads = max(1, int(_first(self.config, "threads", default=4)))
self.port = int(_first(self.config, "port", "server_port", default=0) or 0)
self.startup_timeout = max(
0.1, float(_first(self.config, "startup_timeout", "startup_timeout_seconds", default=30.0))
)
self.request_timeout = max(
0.1, float(_first(self.config, "request_timeout", "request_timeout_seconds", default=600.0))
)
self.connect_timeout = max(
0.05, float(_first(self.config, "connect_timeout", "connect_timeout_seconds", default=2.0))
)
self.stop_timeout = max(
0.1, float(_first(self.config, "stop_timeout", "stop_timeout_seconds", default=5.0))
)
self._lock = threading.RLock()
self._process: Optional[subprocess.Popen] = None
self._client: Optional[AudioCppClient] = None
self._temp_dir: Optional[tempfile.TemporaryDirectory] = None
self._config_path: Optional[Path] = None
self._log_path: Optional[Path] = None
self._log_handle = None
self._closed = False
self._parent_job: Optional[WindowsKillOnCloseJob] = None
@staticmethod
def _resolve_device_index(value: Any) -> int:
text = str(value).strip().lower()
if text in {"auto", "cuda", "hip", "cpu", "vulkan", "metal", ""}:
return 0
if ":" in text:
text = text.rsplit(":", 1)[-1]
try:
return max(0, int(text))
except ValueError as exc:
raise AudioCppProcessStartupError(
f"audio.cpp device must be an integer index, got {value!r}"
) from exc
@property
def process(self) -> Optional[subprocess.Popen]:
return self._process
@property
def client(self) -> AudioCppClient:
if self._client is None:
raise AudioCppProcessError("audio.cpp server process has not started")
return self._client
@property
def config_path(self) -> Optional[Path]:
return self._config_path
@property
def log_path(self) -> Optional[Path]:
return self._log_path
@property
def base_url(self) -> str:
if self.port <= 0:
raise AudioCppProcessError("audio.cpp server port is not allocated")
return f"http://127.0.0.1:{self.port}"
@property
def running(self) -> bool:
return self._process is not None and self._process.poll() is None
def _model_spec_override(self) -> Optional[Path]:
if "model_spec_override" in self.config:
explicit = self.config.get("model_spec_override")
if explicit in (None, "", False):
return None
path = _resolve_existing_path(explicit, "model spec override")
else:
path = (Path(__file__).resolve().parent / "model_specs").resolve()
if not path.is_dir():
raise AudioCppProcessStartupError(
"Bundled audio.cpp release-0.5.1 model specs are missing: " + str(path)
)
return path
def _build_server_config(self) -> Dict[str, Any]:
model: Dict[str, Any] = {
"id": self.model_id,
"family": self.family,
"path": str(self.model_path),
"task": self.task,
"mode": str(_first(self.config, "mode", "run_mode", default="offline")),
"lazy": bool(_first(self.config, "lazy", "lazy_load", default=True)),
"load_options": _option_dict(self.config, "load_options"),
"session_options": _option_dict(self.config, "session_options"),
"default_request_options": _option_dict(self.config, "default_request_options"),
}
for source_key, target_key in (
("config_id", "config"),
("weight_id", "weight"),
("voice_presets", "voice_presets"),
("default_voice_preset", "default_voice_preset"),
("model_busy_timeout_ms", "busy_timeout_ms"),
):
if source_key in self.config and self.config[source_key] is not None:
model[target_key] = self.config[source_key]
server: Dict[str, Any] = {
"host": "127.0.0.1",
"port": self.port,
"cors_origins": "",
"backend": self.backend,
"device": self.device,
"threads": self.threads,
"lazy_load": bool(_first(self.config, "lazy_load", default=True)),
"log_request_body": False,
"max_request_body_bytes": int(
_first(self.config, "max_request_body_bytes", default=2 * 1024 * 1024 * 1024)
),
"busy_timeout_ms": int(_first(self.config, "busy_timeout_ms", default=300000)),
"models": [model],
}
model_spec_override = self._model_spec_override()
if model_spec_override is not None:
server["model_spec_override"] = str(model_spec_override)
return server
def _prepare_files(self) -> None:
temp_root = _first(self.config, "temp_root", "runtime_temp_root")
if temp_root:
Path(str(temp_root)).expanduser().resolve().mkdir(parents=True, exist_ok=True)
self._temp_dir = tempfile.TemporaryDirectory(
prefix="tts_audio_cpp_",
dir=str(Path(str(temp_root)).expanduser().resolve()) if temp_root else None,
)
temp_path = Path(self._temp_dir.name)
self._config_path = temp_path / "server.json"
log_dir = _first(self.config, "log_dir")
if log_dir:
resolved_log_dir = Path(str(log_dir)).expanduser().resolve()
resolved_log_dir.mkdir(parents=True, exist_ok=True)
self._log_path = resolved_log_dir / f"audio_cpp_{self.model_id}_{self.port}.log"
else:
self._log_path = temp_path / "server.log"
with self._config_path.open("w", encoding="utf-8", newline="\n") as handle:
json.dump(self._build_server_config(), handle, ensure_ascii=False, indent=2)
handle.write("\n")
self._log_handle = self._log_path.open("ab", buffering=0)
def _command(self) -> list[str]:
binary_args = _first(self.config, "binary_args", "launcher_args", default=[])
if binary_args is None:
binary_args = []
if isinstance(binary_args, (str, bytes)) or not isinstance(binary_args, Sequence):
raise AudioCppProcessStartupError("audio.cpp binary_args must be a list")
return [
str(self.binary_path),
*[str(item) for item in binary_args],
"--config",
str(self._config_path),
]
def start(self) -> "AudioCppServerProcess":
with self._lock:
if self._closed:
raise AudioCppProcessError("audio.cpp server process launcher is closed")
if self.running:
return self
if self.port <= 0:
self.port = find_free_loopback_port()
try:
self._prepare_files()
env = os.environ.copy()
extra_env = _first(self.config, "environment", "env", default={})
if extra_env:
if not isinstance(extra_env, Mapping):
raise AudioCppProcessStartupError(
"audio.cpp environment must be an object"
)
env.update({str(key): str(value) for key, value in extra_env.items()})
env.setdefault("PYTHONUTF8", "1")
visible_console = bool(self.config.get("show_server_console", False))
creationflags = 0
if os.name == "nt":
creationflags = getattr(
subprocess,
"CREATE_NEW_CONSOLE" if visible_console else "CREATE_NO_WINDOW",
0,
)
self._process = subprocess.Popen(
self._command(),
cwd=str(self.binary_path.parent),
env=env,
stdin=subprocess.DEVNULL,
stdout=None if visible_console else self._log_handle,
stderr=None if visible_console else subprocess.STDOUT,
shell=False,
creationflags=creationflags,
)
if os.name == "nt":
self._parent_job = WindowsKillOnCloseJob()
self._parent_job.assign(int(self._process._handle))
self._client = AudioCppClient(
self.base_url,
connect_timeout=self.connect_timeout,
request_timeout=self.request_timeout,
)
self._wait_until_ready()
return self
except Exception as exc:
log_tail = self.read_log_tail()
self._terminate_exact_process()
self._client = None
self._cleanup_files()
if isinstance(exc, AudioCppProcessStartupError):
raise
suffix = f"\nServer log tail:\n{log_tail}" if log_tail else ""
raise AudioCppProcessStartupError(
f"Failed to start audio.cpp server: {exc}{suffix}"
) from exc
def _wait_until_ready(self) -> None:
deadline = time.monotonic() + self.startup_timeout
last_error: Optional[BaseException] = None
while time.monotonic() < deadline:
if self._process is None or self._process.poll() is not None:
code = self._process.poll() if self._process is not None else "unknown"
log_tail = self.read_log_tail()
suffix = f"\nServer log tail:\n{log_tail}" if log_tail else ""
raise AudioCppProcessStartupError(
f"audio.cpp server exited during startup with code {code}{suffix}"
)
try:
health = self.client.health(timeout=min(self.connect_timeout, 0.5))
status = str(health.get("status", "")).lower()
if status in {"ok", "ready", "healthy"} or health:
models = self.client.models(timeout=min(self.connect_timeout, 1.0))
if any(str(item.get("id")) == self.model_id for item in models):
return
last_error = AudioCppProcessStartupError(
f"audio.cpp server did not register expected model id '{self.model_id}'"
)
except AudioCppClientError as exc:
last_error = exc
time.sleep(0.05)
log_tail = self.read_log_tail()
suffix = f"\nServer log tail:\n{log_tail}" if log_tail else ""
raise AudioCppProcessStartupError(
f"audio.cpp server was not ready after {self.startup_timeout:.1f}s"
+ (f": {last_error}" if last_error else "")
+ suffix
)
def read_log_tail(self, max_bytes: int = 16384) -> str:
path = self._log_path
if path is None or not path.exists():
return ""
try:
if self._log_handle is not None:
self._log_handle.flush()
with path.open("rb") as handle:
size = path.stat().st_size
handle.seek(max(0, size - max(1, int(max_bytes))))
return handle.read().decode("utf-8", errors="replace").strip()
except OSError:
return ""
def _terminate_exact_process(self) -> None:
process = self._process
if process is None:
return
try:
if process.poll() is None:
process.terminate()
try:
process.wait(timeout=self.stop_timeout)
except subprocess.TimeoutExpired:
process.kill()
process.wait(timeout=self.stop_timeout)
finally:
self._process = None
if self._parent_job is not None:
self._parent_job.close()
self._parent_job = None
def _cleanup_files(self) -> None:
if self._log_handle is not None:
try:
self._log_handle.close()
except OSError:
pass
self._log_handle = None
if self._temp_dir is not None:
try:
self._temp_dir.cleanup()
except OSError:
pass
self._temp_dir = None
def close(self) -> None:
with self._lock:
if self._closed:
return
self._closed = True
self._terminate_exact_process()
self._client = None
self._cleanup_files()
stop = close
def __enter__(self) -> "AudioCppServerProcess":
return self.start()
def __exit__(self, exc_type, exc, traceback) -> None:
self.close()
__all__ = [
"AudioCppProcessError",
"AudioCppProcessStartupError",
"AudioCppServerProcess",
"find_free_loopback_port",
"normalize_audio_cpp_task",
]
+609
View File
@@ -0,0 +1,609 @@
"""Resolve workflow and machine-local audio.cpp configuration into one session config."""
from __future__ import annotations
import os
import re
import shutil
import urllib.parse
from pathlib import Path
from typing import Any, Dict, Mapping, Optional, Sequence
from .settings import AudioCppSettings, load_settings
class AudioCppResolutionError(RuntimeError):
"""Raised when an audio.cpp engine configuration cannot be made runnable."""
class _SuiteDownloadProgress:
"""Match UnifiedDownloader's single-line console progress convention."""
def __init__(self) -> None:
self._completed: set[str] = set()
def __call__(self, label: str, downloaded: int, total: Optional[int]) -> None:
if not total or total <= 0:
return
filename = Path(label).name
percent = min(100.0, downloaded * 100.0 / total)
print(f"\r📥 Downloading {filename}: {percent:.1f}%", end="", flush=True)
if downloaded >= total and label not in self._completed:
self._completed.add(label)
print()
def _print_download_block(
title: str,
*,
model: str,
description: str,
repository: str,
target: Path,
size_bytes: Optional[int] = None,
) -> None:
"""Use the same boxed pre-download summary as the Suite engine downloaders."""
print(f"\n{'=' * 60}")
print(f"📦 {title}")
print("=" * 60)
print(f"Model: {model}")
print(f"Description: {description}")
print(f"Repository: {repository}")
print(f"Download size: {_format_download_size(size_bytes)}")
print(f"Target: {target}")
print(f"{'=' * 60}\n")
def _format_download_size(size_bytes: Optional[int]) -> str:
if size_bytes is None or size_bytes < 0:
return "Unknown"
if size_bytes >= 1024**3:
return f"{size_bytes / 1024**3:.2f} GB"
return f"{size_bytes / 1024**2:.1f} MB"
_EXTERNAL_MODES = {
"external",
"external_server",
"existing_server",
"remote_server",
"server",
"connect",
}
_OWNED_MODES = {
"owned",
"owned_process",
"existing_binary",
"managed_binary",
"managed",
"binary",
"local_binary",
}
_TASK_ALIASES = {
"clone": "clon",
"cloning": "clon",
"voice_clone": "clon",
"voice_cloning": "clon",
"voice_design": "vdes",
"design": "vdes",
"voice_conversion": "vc",
"speech_to_speech": "s2s",
"singing_voice_conversion": "svc",
}
def _catalog_module():
from . import catalog
return catalog
def _discovery_module():
from . import discovery
return discovery
def _downloader_module():
from . import downloader
return downloader
def _runtime_installer_module():
from . import runtime_installer
return runtime_installer
def _flatten_config(config: Mapping[str, Any]) -> dict[str, Any]:
if not isinstance(config, Mapping):
raise TypeError("audio.cpp config must be a mapping")
nested = config.get("config")
flattened = dict(nested) if isinstance(nested, Mapping) else {}
flattened.update({key: value for key, value in config.items() if key != "config"})
return flattened
def _has_value(value: Any) -> bool:
if value is None:
return False
if isinstance(value, str):
return bool(value.strip())
if isinstance(value, (list, tuple, set, dict)):
return bool(value)
return True
def _first_value(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
for key in keys:
if key in config and _has_value(config[key]):
return config[key]
return default
def _merge_settings(config: Mapping[str, Any], settings: AudioCppSettings) -> dict[str, Any]:
"""Fill blank/absent workflow fields without replacing explicit values."""
merged = settings.to_mapping()
for key, value in config.items():
if _has_value(value) or key not in merged:
merged[key] = value
return merged
def _normalize_mode(value: Any) -> str:
normalized = re.sub(r"[^a-z0-9]+", "_", str(value or "auto").strip().lower()).strip("_")
if normalized in {"", "auto"}:
return "auto"
if normalized in _EXTERNAL_MODES:
return "external_server"
if normalized in _OWNED_MODES:
return "owned_process"
raise AudioCppResolutionError(f"Unsupported audio.cpp connection mode: {value!r}")
def _canonical_url(value: Any) -> str:
raw = str(value or "").strip()
parsed = urllib.parse.urlsplit(raw)
if parsed.scheme not in {"http", "https"} or not parsed.netloc:
raise AudioCppResolutionError(
"audio.cpp external mode requires a valid HTTP(S) server_url; "
f"received {value!r}"
)
if parsed.query or parsed.fragment:
raise AudioCppResolutionError(
"audio.cpp external server_url must not contain a query string or fragment"
)
return urllib.parse.urlunsplit(
(parsed.scheme.lower(), parsed.netloc, parsed.path.rstrip("/"), "", "")
)
def _path_text(value: Any) -> str:
return os.path.expandvars(os.path.expanduser(os.fspath(value))).strip()
def _existing_path(value: Any, label: str, *, file_only: bool = False) -> Path:
raw = _path_text(value)
if not raw:
raise AudioCppResolutionError(f"Missing audio.cpp {label}")
candidate = Path(raw)
if file_only and not candidate.exists():
located = shutil.which(raw)
if located:
candidate = Path(located)
candidate = candidate.resolve()
if not candidate.exists():
raise AudioCppResolutionError(f"audio.cpp {label} does not exist: {candidate}")
if file_only and not candidate.is_file():
raise AudioCppResolutionError(f"audio.cpp {label} is not a file: {candidate}")
return candidate
def _canonical_model_id(value: Any, fallback: str) -> str:
raw = str(value or fallback or "audio-cpp-model").strip()
cleaned = "".join(char if char.isalnum() or char in "._-" else "-" for char in raw)
return cleaned.strip("-.") or "audio-cpp-model"
def _device_index(value: Any) -> int:
text = str(value if value is not None else 0).strip().lower()
if text in {"", "auto", "cuda", "cpu", "vulkan", "metal", "hip"}:
return 0
if ":" in text:
text = text.rsplit(":", 1)[-1]
try:
index = int(text)
except ValueError as exc:
raise AudioCppResolutionError(
f"audio.cpp device must be a non-negative integer index, got {value!r}"
) from exc
if index < 0:
raise AudioCppResolutionError(
f"audio.cpp device must be a non-negative integer index, got {value!r}"
)
return index
def _as_bool(value: Any, default: bool = False) -> bool:
if value is None:
return default
if isinstance(value, str):
normalized = value.strip().lower()
if normalized in {"1", "true", "yes", "on"}:
return True
if normalized in {"0", "false", "no", "off", ""}:
return False
raise AudioCppResolutionError(f"Invalid boolean value in audio.cpp config: {value!r}")
return bool(value)
def _cuda_available() -> bool:
try:
import torch
return bool(torch.cuda.is_available())
except (ImportError, RuntimeError):
return False
def _resolve_backend(config: Mapping[str, Any], settings: AudioCppSettings, *, owned: bool) -> str:
requested = str(_first_value(config, "backend", default="auto") or "auto").strip().lower()
if owned and requested == "auto" and settings.runtime_backend in {"cpu", "cuda"}:
requested = settings.runtime_backend
if requested == "auto":
return "cuda" if owned and _cuda_available() else ("cpu" if owned else "auto")
supported = {"cuda", "cpu", "vulkan", "metal", "hip"} if owned else {
"cuda",
"cpu",
"vulkan",
"metal",
"hip",
}
if requested not in supported:
raise AudioCppResolutionError(f"Unsupported audio.cpp backend: {requested!r}")
return requested
def _explicit_roots(config: Mapping[str, Any]) -> Optional[Sequence[Any]]:
value = _first_value(config, "model_roots", "model_search_roots")
if value is None:
return None
if isinstance(value, (str, os.PathLike)):
return [value]
if isinstance(value, Sequence):
return list(value)
raise AudioCppResolutionError("audio.cpp model_roots must be a path or list of paths")
def _installed_runtime_binary(runtime_root: Any, backend: str, runtime_api) -> Optional[Path]:
if not _has_value(runtime_root):
return None
root = Path(_path_text(runtime_root)).resolve()
direct = root / "audiocpp_server.exe"
if direct.is_file() and direct.stat().st_size > 0:
return direct.resolve()
if backend not in {"cpu", "cuda"}:
return None
candidate = runtime_api.runtime_install_path(root, backend) / "audiocpp_server.exe"
if candidate.is_file() and candidate.stat().st_size > 0:
return candidate.resolve()
return None
def _external_task(config: Mapping[str, Any]) -> str:
requested = str(
_first_value(config, "requested_task", "task", default="auto") or "auto"
).strip().lower().replace("-", "_").replace(" ", "_")
requested = _TASK_ALIASES.get(requested, requested)
supported = {"auto", "tts", "clon", "vdes", "vc", "s2s", "svc", "asr", "diar"}
if requested not in supported:
raise AudioCppResolutionError(f"Unsupported audio.cpp task: {requested!r}")
return requested
def resolve_audio_cpp_config(
config: Mapping[str, Any],
*,
external: Optional[bool] = None,
) -> dict[str, Any]:
"""Return a complete canonical config without starting an audio.cpp session.
External mode intentionally returns before importing model discovery or either
installer. Owned mode may install only when the corresponding workflow flag
explicitly permits it, and all installs target suite-managed storage.
"""
workflow = _flatten_config(config)
settings = load_settings()
merged = _merge_settings(workflow, settings)
workflow_mode = _normalize_mode(_first_value(workflow, "connection_mode", "source", default="auto"))
workflow_url = _first_value(
workflow, "server_url", "endpoint", "base_url", "external_server_url"
)
merged_url = _first_value(
merged, "server_url", "endpoint", "base_url", "external_server_url"
)
workflow_binary = _first_value(
workflow, "binary_path", "server_binary", "executable_path", "audio_cpp_binary"
)
if external is True:
mode = "external_server"
elif external is False:
mode = "owned_process"
elif workflow_mode != "auto":
mode = workflow_mode
elif _has_value(workflow_url):
# A URL deliberately supplied by the workflow outranks machine defaults.
mode = "external_server"
elif _has_value(workflow_binary):
mode = "owned_process"
elif settings.connection_mode == "external" and _has_value(merged_url):
mode = "external_server"
elif _has_value(settings.executable_path):
mode = "owned_process"
else:
mode = "owned_process"
device_index = _device_index(_first_value(merged, "device_index", "device", default=0))
family = str(_first_value(merged, "family", "model_family", default="") or "").strip()
package_value = str(_first_value(merged, "package_id", default="auto") or "auto").strip()
if mode == "external_server":
endpoint = _canonical_url(merged_url)
package_id = package_value or "auto"
result = dict(merged)
explicit_model_id = _first_value(workflow, "model_id", "server_model_id")
result.update(
{
"connection_mode": "external_server",
"server_url": endpoint,
"external_server_url": endpoint,
"family": family,
"package_id": package_id,
"task": _external_task(workflow),
"backend": _resolve_backend(workflow, settings, owned=False),
"device": device_index,
"device_index": device_index,
"binary_path": "",
"model_path": "",
}
)
if explicit_model_id:
result["model_id"] = _canonical_model_id(explicit_model_id, "")
else:
# Omission is meaningful: the external session will select the
# server's sole /v1/models entry. Its lookup treats blank as an ID.
result.pop("model_id", None)
result.pop("server_model_id", None)
return result
if not family or family.lower() == "auto":
raise AudioCppResolutionError(
"Owned audio.cpp mode requires a model family from the pinned release-0.5.1 catalog"
)
catalog_api = _catalog_module()
try:
catalog = catalog_api.load_catalog()
family_record = catalog.family(family)
package_id = (
family_record.recommended_package_id
if package_value.lower() in {"", "auto"}
else package_value
)
package = catalog.package(package_id)
if package.family != family:
raise AudioCppResolutionError(
f"audio.cpp package {package_id!r} belongs to {package.family!r}, not {family!r}"
)
requested_task = _first_value(
workflow, "requested_task", "task", default=_first_value(merged, "task", default="auto")
)
task = catalog_api.resolve_task(family, package_id, requested=str(requested_task or "auto"))
except AudioCppResolutionError:
raise
except (KeyError, TypeError, ValueError) as exc:
raise AudioCppResolutionError(f"Invalid audio.cpp model selection: {exc}") from exc
backend = _resolve_backend(workflow, settings, owned=True)
discovery_api = _discovery_module()
managed_model_root = discovery_api.default_managed_model_root(settings=settings)
explicit_model = _first_value(workflow, "model_path", "package_path", "gguf_path")
if _has_value(explicit_model):
model_path = _existing_path(explicit_model, "model path")
else:
resolved_model = discovery_api.resolve_model(
package,
_explicit_roots(workflow),
catalog=catalog,
settings=settings,
)
if resolved_model is not None:
model_path = Path(resolved_model.path).resolve()
elif _as_bool(_first_value(workflow, "auto_download_model", default=False)):
downloader_api = _downloader_module()
size_resolver = getattr(downloader_api, "package_download_size", None)
download_size = (
size_resolver(package, catalog=catalog) if callable(size_resolver) else None
)
_print_download_block(
"audio.cpp Model Download",
model=package.display_name,
description=(
f"{family_record.display_name} {package.precision.upper()} "
f"{package.format.upper()} package"
),
repository=package.repo,
target=(Path(managed_model_root) / package.target_directory).resolve(),
size_bytes=download_size,
)
print(f"📥 Downloading {package_id} directly (no cache)")
try:
download_result = downloader_api.install_package(
package,
managed_model_root,
catalog=catalog,
progress=_SuiteDownloadProgress(),
)
except Exception as exc:
raise AudioCppResolutionError(
f"Failed to install audio.cpp package {package_id!r} in managed storage "
f"{managed_model_root}: {exc}"
) from exc
model_path = Path(download_result.path).resolve()
print(f"✅ Downloaded: {model_path}")
else:
searched = discovery_api.resolve_model_roots(
_explicit_roots(workflow), settings=settings
)
raise AudioCppResolutionError(
f"audio.cpp package {package_id!r} is not installed. Searched: "
+ ", ".join(str(Path(root)) for root in searched)
+ ". Provide model_path or enable auto_download_model."
)
dependency_session_options: Dict[str, Any] = {}
try:
from .capabilities import get_package_dependencies
dependencies = get_package_dependencies(package_id)
except (ImportError, KeyError, TypeError, ValueError) as exc:
raise AudioCppResolutionError(
f"Cannot resolve audio.cpp dependencies for {package_id!r}: {exc}"
) from exc
dependency_roots = discovery_api.resolve_model_roots(
_explicit_roots(workflow), settings=settings
)
for dependency in dependencies:
dependency_package = dependency["package"]
dependency_path = discovery_api.find_installed_package(
dependency_package,
dependency_roots,
settings=settings,
)
if dependency_path is None:
if not _as_bool(_first_value(workflow, "auto_download_model", default=False)):
raise AudioCppResolutionError(
f"audio.cpp package {package_id!r} requires {dependency_package.id!r}, "
"which is not installed. Enable auto_download_model to install it."
)
downloader_api = _downloader_module()
_print_download_block(
"audio.cpp Dependency Download",
model=dependency_package.display_name,
description=f"Required by {family_record.display_name}",
repository=dependency_package.repo,
target=(
Path(managed_model_root) / dependency_package.target_directory
).resolve(),
size_bytes=int(dependency["estimated_download_bytes"]),
)
print(f"📥 Downloading {dependency_package.id} directly (no cache)")
try:
dependency_result = downloader_api.install_package(
dependency_package,
managed_model_root,
progress=_SuiteDownloadProgress(),
)
except Exception as exc:
raise AudioCppResolutionError(
f"Failed to install dependency {dependency_package.id!r} for "
f"audio.cpp package {package_id!r}: {exc}"
) from exc
dependency_path = Path(dependency_result.path).resolve()
print(f"✅ Downloaded dependency: {dependency_path}")
dependency_session_options[str(dependency["session_option"])] = str(
Path(dependency_path).resolve()
)
explicit_binary = _first_value(
workflow, "binary_path", "server_binary", "executable_path", "audio_cpp_binary"
)
managed_runtime_root = Path(managed_model_root).expanduser().resolve().parent / "runtime"
if _has_value(explicit_binary):
binary_path = _existing_path(explicit_binary, "server executable", file_only=True)
elif _has_value(settings.executable_path):
binary_path = _existing_path(
settings.executable_path, "configured server executable", file_only=True
)
else:
runtime_api = _runtime_installer_module()
binary_path = _installed_runtime_binary(settings.runtime_root, backend, runtime_api)
if binary_path is None:
binary_path = _installed_runtime_binary(managed_runtime_root, backend, runtime_api)
if binary_path is None and backend == "cpu":
# The official CUDA profile also contains the CPU backend. Reuse it
# before downloading a second executable profile solely for CPU mode.
binary_path = _installed_runtime_binary(settings.runtime_root, "cuda", runtime_api)
if binary_path is None and backend == "cpu":
binary_path = _installed_runtime_binary(managed_runtime_root, "cuda", runtime_api)
if binary_path is None:
if backend not in {"cpu", "cuda"}:
raise AudioCppResolutionError(
f"The pinned managed audio.cpp runtime has no {backend!r} Windows artifact. "
"Provide binary_path for this backend."
)
if not _as_bool(_first_value(workflow, "auto_download_runtime", default=False)):
expected = runtime_api.runtime_install_path(managed_runtime_root, backend)
raise AudioCppResolutionError(
f"audio.cpp {backend} server runtime is not installed at {expected}. "
"Provide binary_path or enable auto_download_runtime."
)
runtime_manifest = runtime_api.get_runtime_manifest(backend)
_print_download_block(
"audio.cpp Runtime Download",
model=f"audio.cpp {runtime_manifest.release_version} ({backend})",
description=(
f"Official Windows {runtime_manifest.profile} runtime, "
f"pinned to {runtime_manifest.release_tag}"
),
repository="0xShug0/audio.cpp",
target=runtime_api.runtime_install_path(managed_runtime_root, backend),
size_bytes=sum(asset.size for asset in runtime_manifest.assets),
)
print(f"📥 Downloading audio.cpp {backend} runtime directly (no cache)")
try:
runtime_result = runtime_api.install_windows_runtime(
managed_runtime_root,
backend,
progress=_SuiteDownloadProgress(),
)
except Exception as exc:
raise AudioCppResolutionError(
f"Failed to install the pinned audio.cpp {backend} balance runtime in "
f"{managed_runtime_root}: {exc}"
) from exc
binary_path = Path(runtime_result.executable).resolve()
print(f"✅ Downloaded: {binary_path}")
result = dict(merged)
session_options = dict(result.get("session_options") or {})
session_options.update(dependency_session_options)
result.update(
{
"connection_mode": "owned_process",
"server_url": "",
"external_server_url": "",
"family": family,
"package_id": package_id,
"model_id": _canonical_model_id(
_first_value(workflow, "model_id", "server_model_id"), package_id
),
"task": task,
"backend": backend,
"device": device_index,
"device_index": device_index,
"binary_path": str(binary_path),
"model_path": str(model_path),
"session_options": session_options,
}
)
return result
__all__ = ["AudioCppResolutionError", "resolve_audio_cpp_config"]
+252
View File
@@ -0,0 +1,252 @@
"""Verified Windows audio.cpp release-0.5.1 runtime installer."""
from __future__ import annotations
import hashlib
import os
import shutil
import stat
import sys
import tempfile
import uuid
import zipfile
from dataclasses import dataclass
from pathlib import Path, PurePosixPath
from typing import Callable, Mapping, Optional, Sequence, Union
from .catalog import AUDIO_CPP_RELEASE_COMMIT, AUDIO_CPP_RELEASE_TAG, AUDIO_CPP_RELEASE_VERSION
from .downloader import AudioCppDownloadError, ProgressCallback, download_url_to_path
class RuntimeInstallError(RuntimeError):
"""Raised when a managed audio.cpp runtime cannot be verified or installed."""
@dataclass(frozen=True)
class RuntimeAsset:
filename: str
url: str
size: int
sha256: str
@dataclass(frozen=True)
class RuntimeManifest:
backend: str
assets: tuple[RuntimeAsset, ...]
required_files: tuple[str, ...]
release_version: str = AUDIO_CPP_RELEASE_VERSION
release_tag: str = AUDIO_CPP_RELEASE_TAG
release_commit: str = AUDIO_CPP_RELEASE_COMMIT
profile: str = "balance"
platform: str = "windows"
_RELEASE_URL = "https://github.com/0xShug0/audio.cpp/releases/download/release-0.5.1"
WINDOWS_RUNTIME_MANIFESTS: Mapping[str, RuntimeManifest] = {
"cpu": RuntimeManifest(
backend="cpu",
assets=(
RuntimeAsset(
filename="audiocpp-windows-cpu-balance-238ab6a9.zip",
url=f"{_RELEASE_URL}/audiocpp-windows-cpu-balance-238ab6a9.zip",
size=11_435_334,
sha256="c9db54d75becfc9dfa6930d469b6f201f0f3919e36dfa7f488d0ee754ab5fe23",
),
),
required_files=("audiocpp_server.exe", "audiocpp_cli.exe"),
),
"cuda": RuntimeManifest(
backend="cuda",
assets=(
RuntimeAsset(
filename="audiocpp-windows-cuda-balance-238ab6a9.zip",
url=f"{_RELEASE_URL}/audiocpp-windows-cuda-balance-238ab6a9.zip",
size=248_519_503,
sha256="7e20f1fa984960b327700365a9a8434dc8effa0fcaa8d871c6cf5a35bf77b2a7",
),
RuntimeAsset(
filename="audiocpp-windows-cuda-runtime.zip",
url=f"{_RELEASE_URL}/audiocpp-windows-cuda-runtime.zip",
size=575_505_446,
sha256="46016655aff8f050806d81efd0fe256c15b86527935bfb3896208d4cac6b5ff8",
),
),
required_files=(
"audiocpp_server.exe",
"audiocpp_cli.exe",
"cublas64_13.dll",
"cublasLt64_13.dll",
"cufft64_12.dll",
),
),
}
@dataclass(frozen=True)
class RuntimeInstallResult:
manifest: RuntimeManifest
path: Path
executable: Path
already_present: bool = False
def get_runtime_manifest(backend: str) -> RuntimeManifest:
normalized = str(backend).strip().lower()
try:
return WINDOWS_RUNTIME_MANIFESTS[normalized]
except KeyError as exc:
raise RuntimeInstallError("audio.cpp runtime backend must be 'cpu' or 'cuda'") from exc
def runtime_install_path(runtime_root: Union[str, Path], backend: str) -> Path:
manifest = get_runtime_manifest(backend)
return (
Path(runtime_root).expanduser()
/ f"release-{manifest.release_version}"
/ f"windows-{manifest.backend}-{manifest.profile}"
)
def runtime_is_complete(path: Union[str, Path], manifest: RuntimeManifest) -> bool:
root = Path(path)
return root.is_dir() and all(
(root / relative).is_file() and (root / relative).stat().st_size > 0
for relative in manifest.required_files
)
def _sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def _safe_member_path(member_name: str) -> Path:
normalized = member_name.replace("\\", "/")
path = PurePosixPath(normalized)
if (
not normalized
or path.is_absolute()
or ".." in path.parts
or (path.parts and ":" in path.parts[0])
):
raise RuntimeInstallError(f"Unsafe path in runtime archive: {member_name!r}")
return Path(*path.parts)
def _extract_zip(archive: Path, destination: Path) -> None:
try:
with zipfile.ZipFile(archive) as bundle:
for info in bundle.infolist():
relative = _safe_member_path(info.filename)
unix_mode = info.external_attr >> 16
if stat.S_ISLNK(unix_mode):
raise RuntimeInstallError(f"Runtime archive contains a symlink: {info.filename}")
target = destination / relative
if info.is_dir():
target.mkdir(parents=True, exist_ok=True)
continue
target.parent.mkdir(parents=True, exist_ok=True)
with bundle.open(info, "r") as source, target.open("wb") as output:
shutil.copyfileobj(source, output, length=1024 * 1024)
except (OSError, zipfile.BadZipFile) as exc:
raise RuntimeInstallError(f"Cannot extract runtime archive {archive.name}: {exc}") from exc
def _remove_path(path: Path) -> None:
if path.is_symlink() or path.is_file():
path.unlink(missing_ok=True)
elif path.is_dir():
shutil.rmtree(path)
def _publish_directory(staging: Path, target: Path, overwrite: bool) -> None:
if not target.exists() and not target.is_symlink():
staging.rename(target)
return
if not overwrite:
raise RuntimeInstallError(f"Runtime target exists but is incomplete: {target}")
backup = target.with_name(f".{target.name}.{uuid.uuid4().hex}.backup")
target.rename(backup)
try:
staging.rename(target)
except BaseException:
if not target.exists() and backup.exists():
backup.rename(target)
raise
_remove_path(backup)
def install_windows_runtime(
runtime_root: Union[str, Path],
backend: str,
*,
overwrite: bool = False,
manifest: Optional[RuntimeManifest] = None,
platform_name: Optional[str] = None,
timeout: int = 600,
opener=None,
progress: Optional[ProgressCallback] = None,
) -> RuntimeInstallResult:
"""Download, hash, extract, validate, and atomically publish a runtime."""
platform_value = (platform_name or sys.platform).lower()
if platform_value not in {"win32", "windows"}:
raise RuntimeInstallError("Managed audio.cpp release binaries are currently Windows-only")
selected = manifest or get_runtime_manifest(backend)
if selected.backend != str(backend).strip().lower():
raise RuntimeInstallError("Runtime manifest backend does not match the requested backend")
target = runtime_install_path(runtime_root, selected.backend)
if runtime_is_complete(target, selected) and not overwrite:
return RuntimeInstallResult(
selected, target, target / "audiocpp_server.exe", already_present=True
)
if (target.exists() or target.is_symlink()) and not overwrite:
raise RuntimeInstallError(f"Runtime target exists but is incomplete: {target}")
target.parent.mkdir(parents=True, exist_ok=True)
work = Path(tempfile.mkdtemp(prefix=f".{target.name}.", suffix=".staging", dir=target.parent))
archives = work / "archives"
payload = work / "payload"
archives.mkdir()
payload.mkdir()
try:
for asset in selected.assets:
archive = archives / asset.filename
try:
downloaded = download_url_to_path(
asset.url,
archive,
timeout=timeout,
opener=opener,
progress=progress,
progress_label=asset.filename,
)
except AudioCppDownloadError as exc:
raise RuntimeInstallError(str(exc)) from exc
if downloaded != asset.size:
raise RuntimeInstallError(
f"Runtime asset {asset.filename} has size {downloaded}; expected {asset.size}"
)
actual_hash = _sha256(archive)
if actual_hash.lower() != asset.sha256.lower():
raise RuntimeInstallError(
f"SHA256 mismatch for {asset.filename}: expected {asset.sha256}, got {actual_hash}"
)
_extract_zip(archive, payload)
if not runtime_is_complete(payload, selected):
missing = [
item
for item in selected.required_files
if not (payload / item).is_file() or (payload / item).stat().st_size == 0
]
raise RuntimeInstallError(f"Runtime staging validation failed; missing: {missing}")
_publish_directory(payload, target, overwrite=overwrite)
return RuntimeInstallResult(selected, target, target / "audiocpp_server.exe")
finally:
_remove_path(work)
+626
View File
@@ -0,0 +1,626 @@
"""Keyed audio.cpp sessions and ComfyUI lifecycle integration."""
from __future__ import annotations
import atexit
import ipaddress
import json
import os
import re
import sys
import threading
import time
import urllib.parse
import uuid
from pathlib import Path
from typing import Any, Dict, Mapping, Optional
from .client import (
AudioCppClient,
AudioCppConnectionError,
AudioCppTaskResult,
AudioCppTimeoutError,
)
from .lifecycle import AudioCppRuntimeProxy
from .process import AudioCppServerProcess, normalize_audio_cpp_task
from .resolver import resolve_audio_cpp_config
def _warn(message: str, exc: Optional[BaseException] = None) -> None:
"""Emit diagnostics without assuming a UTF-8 Windows console."""
text = f"WARNING: {message}"
if exc is not None:
text += f": {exc}"
encoding = getattr(sys.stderr, "encoding", None) or "ascii"
try:
text = text.encode(encoding, errors="replace").decode(encoding, errors="replace")
except LookupError:
text = text.encode("ascii", errors="replace").decode("ascii")
print(text, file=sys.stderr)
def _flatten_config(config: Mapping[str, Any]) -> Dict[str, Any]:
if not isinstance(config, Mapping):
raise TypeError("audio.cpp config must be a mapping")
nested = config.get("config")
flattened: Dict[str, Any] = dict(nested) if isinstance(nested, Mapping) else {}
flattened.update({key: value for key, value in config.items() if key != "config"})
return flattened
def _first(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
for key in keys:
if key in config and config[key] is not None:
return config[key]
return default
def _connection_mode(config: Mapping[str, Any]) -> str:
raw = str(_first(config, "connection_mode", "source", default="auto"))
normalized = re.sub(r"[^a-z0-9]+", "_", raw.strip().lower()).strip("_")
external_aliases = {
"existing_server",
"external_server",
"server",
"remote_server",
"connect",
}
owned_aliases = {
"owned_process",
"existing_binary",
"managed_binary",
"managed",
"binary",
"local_binary",
}
if normalized in external_aliases:
return "external_server"
if normalized in owned_aliases:
return "owned_process"
if normalized not in {"", "auto"}:
raise ValueError(f"Unsupported audio.cpp connection mode: {raw}")
endpoint = _first(config, "server_url", "endpoint", "base_url")
return "external_server" if endpoint else "owned_process"
def _is_loopback_url(url: str) -> bool:
parsed = urllib.parse.urlsplit(url)
host = (parsed.hostname or "").strip().lower()
if host == "localhost":
return True
try:
return ipaddress.ip_address(host).is_loopback
except ValueError:
return False
def _canonical_url(url: Any) -> str:
parsed = urllib.parse.urlsplit(str(url or "").strip())
if parsed.scheme not in {"http", "https"} or not parsed.netloc:
raise ValueError(f"Invalid audio.cpp external server URL: {url!r}")
if parsed.query or parsed.fragment:
raise ValueError("audio.cpp external server URL must not contain a query or fragment")
return urllib.parse.urlunsplit(
(parsed.scheme, parsed.netloc, parsed.path.rstrip("/"), "", "")
)
def _safe_model_id(config: Mapping[str, Any]) -> str:
raw = _first(
config,
"model_id",
"server_model_id",
"package_id",
default=_first(config, "family", "model_family", default="audio-cpp-model"),
)
value = str(raw or "audio-cpp-model").strip()
cleaned = "".join(char if char.isalnum() or char in "._-" else "-" for char in value)
return cleaned.strip("-.") or "audio-cpp-model"
def _jsonable(value: Any) -> Any:
if isinstance(value, Mapping):
return {str(key): _jsonable(item) for key, item in sorted(value.items(), key=lambda item: str(item[0]))}
if isinstance(value, (list, tuple)):
return [_jsonable(item) for item in value]
if isinstance(value, Path):
return str(value.expanduser().resolve())
if isinstance(value, (str, int, float, bool)) or value is None:
return value
return repr(value)
def _session_key(config: Mapping[str, Any], mode: str, model_id: str, endpoint: str = "") -> str:
if mode == "external_server":
identity = {
"mode": mode,
"endpoint": endpoint,
"model_id": model_id,
}
else:
identity_keys = (
"binary_path",
"server_binary",
"executable_path",
"audio_cpp_binary",
"binary_args",
"model_path",
"package_path",
"gguf_path",
"family",
"model_family",
"task",
"backend",
"device",
"device_index",
"threads",
"load_options",
"session_options",
"default_request_options",
"config_id",
"weight_id",
"model_spec_override",
"show_server_console",
)
identity = {"mode": mode, "model_id": model_id}
for key in identity_keys:
if key in config:
identity[key] = config[key]
return json.dumps(_jsonable(identity), ensure_ascii=True, sort_keys=True, separators=(",", ":"))
def _normalize_request_paths(request: Mapping[str, Any]) -> Dict[str, Any]:
normalized = dict(request)
path_fields = {
"voice_ref",
"audio_path",
"source_audio",
"target_voice",
"prosody_ref",
"style_ref",
}
for key in path_fields:
value = normalized.get(key)
if isinstance(value, os.PathLike) or (isinstance(value, str) and value.strip()):
normalized[key] = str(
Path(os.path.expandvars(os.path.expanduser(str(value)))).resolve()
)
return normalized
class AudioCppSession:
"""Persistent external connection or restartable suite-owned audio.cpp server."""
def __init__(
self,
config: Mapping[str, Any],
*,
owned: bool,
model_id: str,
client: Optional[AudioCppClient] = None,
endpoint: str = "",
model_metadata: Optional[Mapping[str, Any]] = None,
) -> None:
self.config = dict(config)
self.owned = bool(owned)
self.model_id = str(model_id)
self.endpoint = endpoint
self.model_metadata = dict(model_metadata or {})
self.family = str(
_first(
self.model_metadata,
"family",
default=_first(self.config, "family", "model_family", default=""),
)
or ""
)
self.task = normalize_audio_cpp_task(
_first(
self.model_metadata,
"task",
default=_first(self.config, "task", default="tts"),
)
)
self.unload_models_supported = False
self._client = client
self._process: Optional[AudioCppServerProcess] = None
self._lock = threading.RLock()
self._closed = False
self._model_ready_reported = False
self.ui_session_id = uuid.uuid4().hex
self._proxy = AudioCppRuntimeProxy(self) if self.owned else None
@property
def process(self) -> Optional[AudioCppServerProcess]:
return self._process
@property
def running(self) -> bool:
if not self.owned:
return not self._closed
return self._process is not None and self._process.running
@property
def proxy(self) -> Optional[AudioCppRuntimeProxy]:
return self._proxy
def _probe_opt_in_features(self) -> None:
if not bool(self.config.get("probe_unload_models", False)) or self._client is None:
return
try:
self.unload_models_supported = self._client.supports_feature("unload_models")
except Exception as exc:
_warn("audio.cpp unload_models feature probe failed", exc)
self.unload_models_supported = False
def _start_owned_runtime(self) -> None:
if not self.owned:
return
with _RUNTIME_START_LOCK:
if self._process is not None and self._process.running:
return
_stop_conflicting_audio_cpp_sessions(self)
if str(self.config.get("backend", "cuda")).lower() in {"cuda", "hip", "auto"}:
_clear_conflicting_suite_tts_models()
process_config = dict(self.config)
process_config["model_id"] = self.model_id
family = self.family or str(self.config.get("family", "unknown"))
backend = str(self.config.get("backend", "auto"))
print(
f"🚀 audio.cpp: Starting {backend} server for {family} "
f"('{self.model_id}')..."
)
started = time.monotonic()
process = AudioCppServerProcess(process_config)
process.start()
self._process = process
self._client = process.client
self.endpoint = process.base_url
self._probe_opt_in_features()
if self._proxy is not None:
self._proxy.register()
print(
f"✅ audio.cpp: Server ready at {self.endpoint} "
f"({time.monotonic() - started:.2f}s); model will load on first request"
)
def _ensure_client(self) -> AudioCppClient:
if self._closed:
raise RuntimeError("audio.cpp session is closed")
if self.owned:
if self._process is None or not self._process.running or self._client is None:
if self._proxy is not None:
self._proxy.unregister()
if self._process is not None:
self._process.close()
self._process = None
self._client = None
self._start_owned_runtime()
if self._client is None:
raise RuntimeError("audio.cpp session has no HTTP client")
return self._client
def _restart_after_transport_failure(self) -> AudioCppClient:
if not self.owned:
raise RuntimeError("Cannot restart an external audio.cpp server")
self._stop_owned_runtime()
self._start_owned_runtime()
if self._client is None:
raise RuntimeError("audio.cpp owned runtime restart did not create a client")
return self._client
def restart_owned_runtime(self) -> None:
"""Recreate an owned server when an upstream session cannot be reused safely."""
if not self.owned:
raise RuntimeError("Cannot restart an external audio.cpp server")
with self._lock:
self._stop_owned_runtime()
self._start_owned_runtime()
def run(self, request: Mapping[str, Any]) -> AudioCppTaskResult:
if not isinstance(request, Mapping):
raise TypeError("audio.cpp task request must be a mapping")
normalized_request = _normalize_request_paths(request)
timeout = float(_first(self.config, "request_timeout", "request_timeout_seconds", default=600.0))
with self._lock:
client = self._ensure_client()
first_request = not self._model_ready_reported
started = time.monotonic()
if first_request:
action = "Loading model" if self.owned else "Sending first request to model"
print(f"⏳ audio.cpp: {action} '{self.model_id}'...")
try:
result = client.run_task(self.model_id, normalized_request, timeout=timeout)
except AudioCppConnectionError:
if not self.owned:
raise
client = self._restart_after_transport_failure()
result = client.run_task(self.model_id, normalized_request, timeout=timeout)
except AudioCppTimeoutError:
# A live server may still be executing after the client times out.
# Only restart when the exact child has actually exited.
if not self.owned or (self._process is not None and self._process.running):
raise
client = self._restart_after_transport_failure()
result = client.run_task(self.model_id, normalized_request, timeout=timeout)
if first_request:
self._model_ready_reported = True
print(
f"✅ audio.cpp: Model '{self.model_id}' loaded; first {self.task} request completed "
f"in {time.monotonic() - started:.2f}s"
)
return result
def voices(self) -> list[str]:
timeout = float(_first(self.config, "connect_timeout", "connect_timeout_seconds", default=5.0))
with self._lock:
client = self._ensure_client()
try:
return client.voices(self.model_id, timeout=timeout)
except AudioCppConnectionError:
if not self.owned:
raise
return self._restart_after_transport_failure().voices(
self.model_id, timeout=timeout
)
def _stop_owned_runtime(self, *, unregister: bool = True) -> None:
if not self.owned:
return
with self._lock:
process = self._process
self._process = None
self._client = None
self._model_ready_reported = False
if unregister and self._proxy is not None:
self._proxy.unregister()
if process is not None:
process.close()
def close(self) -> None:
with self._lock:
if self._closed:
return
self._closed = True
if self.owned:
self._stop_owned_runtime()
else:
# External servers are never unloaded, reconfigured, or terminated.
self._client = None
_SESSIONS: Dict[str, AudioCppSession] = {}
_SESSIONS_LOCK = threading.RLock()
_RUNTIME_START_LOCK = threading.RLock()
def _stop_conflicting_audio_cpp_sessions(active: AudioCppSession) -> None:
if str(active.config.get("backend", "cuda")).lower() not in {"cuda", "hip", "auto"}:
return
with _SESSIONS_LOCK:
conflicts = [
session
for session in _SESSIONS.values()
if session is not active
and session.owned
and session.running
and str(session.config.get("backend", "cuda")).lower()
in {"cuda", "hip", "auto"}
]
for session in conflicts:
session._stop_owned_runtime()
def _clear_conflicting_suite_tts_models() -> None:
"""Clear known suite-managed TTS resources only when their modules are already live."""
interface_module = sys.modules.get("utils.models.unified_model_interface")
interface = getattr(interface_module, "unified_model_interface", None)
if interface is not None:
try:
isolated = getattr(interface, "_isolated_model_cache", None)
remover = getattr(interface, "_remove_isolated_model", None)
if isinstance(isolated, dict) and callable(remover):
for cache_key in list(isolated):
if "_tts_" in cache_key:
remover(cache_key)
except Exception as exc:
_warn("Could not clear a conflicting isolated TTS runtime", exc)
wrapper_module = sys.modules.get("utils.models.comfyui_model_wrapper")
manager = getattr(wrapper_module, "tts_model_manager", None)
cache = getattr(manager, "_model_cache", None)
remover = getattr(manager, "remove_model", None)
if isinstance(cache, dict) and callable(remover):
try:
for cache_key, wrapper in list(cache.items()):
model_info = getattr(wrapper, "model_info", None)
if getattr(model_info, "model_type", None) == "tts":
remover(cache_key)
except Exception as exc:
_warn("Could not clear a conflicting embedded TTS model", exc)
def _external_client(
config: Mapping[str, Any],
) -> tuple[AudioCppClient, str, str, Dict[str, Any]]:
endpoint = _canonical_url(_first(config, "server_url", "endpoint", "base_url"))
if not _is_loopback_url(endpoint) and not bool(config.get("allow_remote_server", False)):
raise ValueError(
"External audio.cpp servers must use loopback by default. "
"Set allow_remote_server only when transport security and path access are understood."
)
client = AudioCppClient(
endpoint,
connect_timeout=float(
_first(config, "connect_timeout", "connect_timeout_seconds", default=5.0)
),
request_timeout=float(
_first(config, "request_timeout", "request_timeout_seconds", default=600.0)
),
)
models = client.models()
requested_id = _first(config, "model_id", "server_model_id")
available_models = [dict(item) for item in models if item.get("id") is not None]
available_ids = [str(item["id"]) for item in available_models]
if requested_id is not None:
model_id = str(requested_id)
if model_id not in available_ids:
raise ValueError(
f"External audio.cpp server does not expose model '{model_id}'. "
f"Available: {', '.join(available_ids) or '(none)'}"
)
elif len(available_ids) == 1:
model_id = available_ids[0]
elif not available_ids:
raise ValueError("External audio.cpp server does not expose any configured models")
else:
raise ValueError(
"External audio.cpp server exposes multiple models; select model_id explicitly"
)
selected_metadata = next(
(item for item in available_models if str(item.get("id")) == model_id),
{"id": model_id},
)
return client, endpoint, model_id, selected_metadata
def get_audio_cpp_session(config: Mapping[str, Any]) -> AudioCppSession:
"""Return a keyed persistent audio.cpp session for a loose engine config dict."""
flattened = resolve_audio_cpp_config(_flatten_config(config))
mode = _connection_mode(flattened)
if mode == "external_server":
endpoint_hint = _canonical_url(
_first(flattened, "server_url", "endpoint", "base_url")
)
model_hint = _first(flattened, "model_id", "server_model_id")
auto_key = None
if isinstance(model_hint, str) and model_hint.strip():
hinted_key = _session_key(flattened, mode, model_hint.strip(), endpoint_hint)
with _SESSIONS_LOCK:
existing = _SESSIONS.get(hinted_key)
if existing is not None and not existing._closed:
return existing
else:
# An omitted model id means "select the server's sole model". Cache
# that resolution per endpoint so every text chunk does not repeat
# /v1/models before reaching the already-persistent session.
auto_key = _session_key(flattened, mode, "", endpoint_hint)
with _SESSIONS_LOCK:
existing = _SESSIONS.get(auto_key)
if existing is not None and not existing._closed:
return existing
client, endpoint, model_id, model_metadata = _external_client(flattened)
key = _session_key(flattened, mode, model_id, endpoint)
with _SESSIONS_LOCK:
existing = _SESSIONS.get(key)
if existing is not None and not existing._closed:
if auto_key is not None:
_SESSIONS[auto_key] = existing
return existing
session = AudioCppSession(
flattened,
owned=False,
model_id=model_id,
client=client,
endpoint=endpoint,
model_metadata=model_metadata,
)
session._probe_opt_in_features()
_SESSIONS[key] = session
if auto_key is not None:
_SESSIONS[auto_key] = session
return session
model_id = _safe_model_id(flattened)
key = _session_key(flattened, mode, model_id)
with _SESSIONS_LOCK:
existing = _SESSIONS.get(key)
if existing is not None and not existing._closed:
return existing
session = AudioCppSession(flattened, owned=True, model_id=model_id)
_SESSIONS[key] = session
return session
def close_all_audio_cpp_sessions() -> None:
with _SESSIONS_LOCK:
sessions = list({id(session): session for session in _SESSIONS.values()}.values())
_SESSIONS.clear()
for session in sessions:
try:
session.close()
except Exception as exc:
_warn("Failed to close audio.cpp session during shutdown", exc)
def audio_cpp_session_statuses() -> list[Dict[str, Any]]:
"""Return a path-free, side-effect-free snapshot for the frontend indicator."""
with _SESSIONS_LOCK:
sessions = list({id(session): session for session in _SESSIONS.values()}.values())
statuses = []
for session in sessions:
if session._closed:
continue
if session.owned:
if session.running:
state = "model_ready" if session._model_ready_reported else "server_ready"
else:
state = "configured"
else:
state = "model_ready" if session._model_ready_reported else "server_ready"
statuses.append(
{
"session_id": session.ui_session_id,
"state": state,
"owned": session.owned,
"family": session.family,
"model_id": session.model_id,
"endpoint": session.endpoint if not session.owned else "",
"pid": session.process.process.pid if session.owned and session.running else None,
**_process_memory_status(
session.process.process.pid if session.owned and session.running else None
),
}
)
return statuses
def _process_memory_status(pid: Optional[int]) -> Dict[str, Optional[int]]:
if not pid:
return {"working_set_bytes": None, "private_bytes": None}
try:
import psutil
info = psutil.Process(pid).memory_info()
return {
"working_set_bytes": int(info.rss),
"private_bytes": int(getattr(info, "private", info.vms)),
}
except Exception:
return {"working_set_bytes": None, "private_bytes": None}
def stop_owned_audio_cpp_session(session_id: str) -> bool:
"""Stop, but retain, an exact Suite-owned session for lazy restart."""
with _SESSIONS_LOCK:
sessions = list({id(session): session for session in _SESSIONS.values()}.values())
session = next((item for item in sessions if item.ui_session_id == session_id), None)
if session is None:
return False
if not session.owned:
raise PermissionError("External audio.cpp servers cannot be stopped by the Suite")
session._stop_owned_runtime()
return True
atexit.register(close_all_audio_cpp_sessions)
__all__ = [
"AudioCppRuntimeProxy",
"AudioCppSession",
"audio_cpp_session_statuses",
"close_all_audio_cpp_sessions",
"get_audio_cpp_session",
"stop_owned_audio_cpp_session",
]
+175
View File
@@ -0,0 +1,175 @@
"""Machine-local audio.cpp settings stored outside workflow JSON."""
from __future__ import annotations
import json
import os
import tempfile
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any, Dict, Mapping, Optional, Tuple
SETTINGS_SCHEMA_VERSION = 1
SETTINGS_FILENAME = "settings.json"
class SettingsError(ValueError):
"""Raised for invalid audio.cpp machine-local settings."""
@dataclass(frozen=True)
class AudioCppSettings:
schema_version: int = SETTINGS_SCHEMA_VERSION
connection_mode: str = "managed"
external_server_url: str = ""
executable_path: str = ""
model_roots: Tuple[str, ...] = ()
managed_model_root: str = ""
runtime_root: str = ""
runtime_backend: str = "auto"
host: str = "127.0.0.1"
port: int = 0
extras: Mapping[str, Any] = field(default_factory=dict, repr=False, compare=False)
@classmethod
def from_mapping(cls, values: Mapping[str, Any]) -> "AudioCppSettings":
if not isinstance(values, Mapping):
raise SettingsError("audio.cpp settings must be a JSON object")
known = {
"schema_version",
"connection_mode",
"external_server_url",
"executable_path",
"model_roots",
"managed_model_root",
"runtime_root",
"runtime_backend",
"host",
"port",
}
roots = values.get("model_roots", ())
if roots is None:
roots = ()
if not isinstance(roots, (list, tuple)) or not all(isinstance(item, str) for item in roots):
raise SettingsError("model_roots must be a list of paths")
mode = str(values.get("connection_mode", "managed")).strip().lower()
if mode not in {"managed", "external"}:
raise SettingsError("connection_mode must be 'managed' or 'external'")
backend = str(values.get("runtime_backend", "auto")).strip().lower()
if backend not in {"auto", "cpu", "cuda"}:
raise SettingsError("runtime_backend must be 'auto', 'cpu', or 'cuda'")
try:
port = int(values.get("port", 0))
except (TypeError, ValueError) as exc:
raise SettingsError("port must be an integer") from exc
if not 0 <= port <= 65535:
raise SettingsError("port must be between 0 and 65535")
schema_version = int(values.get("schema_version", SETTINGS_SCHEMA_VERSION))
if schema_version > SETTINGS_SCHEMA_VERSION:
raise SettingsError(
f"Unsupported audio.cpp settings schema {schema_version}; "
f"maximum is {SETTINGS_SCHEMA_VERSION}"
)
return cls(
schema_version=SETTINGS_SCHEMA_VERSION,
connection_mode=mode,
external_server_url=str(values.get("external_server_url", "")).strip(),
executable_path=str(values.get("executable_path", "")).strip(),
model_roots=tuple(item.strip() for item in roots if item.strip()),
managed_model_root=str(values.get("managed_model_root", "")).strip(),
runtime_root=str(values.get("runtime_root", "")).strip(),
runtime_backend=backend,
host=str(values.get("host", "127.0.0.1")).strip() or "127.0.0.1",
port=port,
extras={key: value for key, value in values.items() if key not in known},
)
def to_mapping(self) -> Dict[str, Any]:
values = dict(self.extras)
serialized = asdict(self)
serialized.pop("extras", None)
serialized["model_roots"] = list(self.model_roots)
values.update(serialized)
return values
def _import_folder_paths():
try:
import folder_paths # type: ignore
return folder_paths
except (ImportError, RuntimeError):
return None
def _fallback_settings_directory() -> Path:
if os.name == "nt" and os.environ.get("LOCALAPPDATA"):
return Path(os.environ["LOCALAPPDATA"]) / "TTS Audio Suite" / "audio_cpp"
if os.environ.get("XDG_CONFIG_HOME"):
return Path(os.environ["XDG_CONFIG_HOME"]) / "tts_audio_suite" / "audio_cpp"
return Path.home() / ".config" / "tts_audio_suite" / "audio_cpp"
def get_settings_path(folder_paths_module=None) -> Path:
"""Resolve settings below ComfyUI's internal user directory when available."""
module = folder_paths_module if folder_paths_module is not None else _import_folder_paths()
if module is not None and hasattr(module, "get_system_user_directory"):
try:
base = Path(module.get_system_user_directory("tts_audio_suite"))
return base / "audio_cpp" / SETTINGS_FILENAME
except (OSError, TypeError, ValueError):
pass
return _fallback_settings_directory() / SETTINGS_FILENAME
def load_settings(
path: Optional[Path] = None,
*,
folder_paths_module=None,
strict: bool = False,
) -> AudioCppSettings:
settings_path = Path(path) if path is not None else get_settings_path(folder_paths_module)
if not settings_path.is_file():
return AudioCppSettings()
try:
values = json.loads(settings_path.read_text(encoding="utf-8"))
return AudioCppSettings.from_mapping(values)
except (OSError, json.JSONDecodeError, SettingsError, TypeError, ValueError):
if strict:
raise
# A damaged local preference file must not stop ComfyUI from loading.
return AudioCppSettings()
def save_settings(
settings: AudioCppSettings,
path: Optional[Path] = None,
*,
folder_paths_module=None,
) -> Path:
"""Atomically write settings in the same directory as the final file."""
if not isinstance(settings, AudioCppSettings):
settings = AudioCppSettings.from_mapping(settings) # type: ignore[arg-type]
settings_path = Path(path) if path is not None else get_settings_path(folder_paths_module)
settings_path.parent.mkdir(parents=True, exist_ok=True)
descriptor, temporary_name = tempfile.mkstemp(
prefix=f".{settings_path.name}.", suffix=".tmp", dir=settings_path.parent
)
temporary_path = Path(temporary_name)
try:
with os.fdopen(descriptor, "w", encoding="utf-8", newline="\n") as handle:
json.dump(settings.to_mapping(), handle, indent=2, sort_keys=True, ensure_ascii=False)
handle.write("\n")
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary_path, settings_path)
except BaseException:
try:
temporary_path.unlink(missing_ok=True)
except OSError:
pass
raise
return settings_path
+73
View File
@@ -0,0 +1,73 @@
"""Windows Job Object ownership for suite-launched native servers."""
from __future__ import annotations
import os
class WindowsKillOnCloseJob:
"""Kill assigned children when the owning Python process loses this handle."""
def __init__(self) -> None:
self.handle = None
def assign(self, process_handle: int) -> None:
if os.name != "nt":
return
import ctypes
from ctypes import wintypes
class IO_COUNTERS(ctypes.Structure):
_fields_ = [(name, ctypes.c_ulonglong) for name in (
"ReadOperationCount", "WriteOperationCount", "OtherOperationCount",
"ReadTransferCount", "WriteTransferCount", "OtherTransferCount",
)]
class JOBOBJECT_BASIC_LIMIT_INFORMATION(ctypes.Structure):
_fields_ = [
("PerProcessUserTimeLimit", ctypes.c_longlong),
("PerJobUserTimeLimit", ctypes.c_longlong),
("LimitFlags", wintypes.DWORD),
("MinimumWorkingSetSize", ctypes.c_size_t),
("MaximumWorkingSetSize", ctypes.c_size_t),
("ActiveProcessLimit", wintypes.DWORD),
("Affinity", ctypes.c_size_t),
("PriorityClass", wintypes.DWORD),
("SchedulingClass", wintypes.DWORD),
]
class JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure):
_fields_ = [
("BasicLimitInformation", JOBOBJECT_BASIC_LIMIT_INFORMATION),
("IoInfo", IO_COUNTERS),
("ProcessMemoryLimit", ctypes.c_size_t),
("JobMemoryLimit", ctypes.c_size_t),
("PeakProcessMemoryUsed", ctypes.c_size_t),
("PeakJobMemoryUsed", ctypes.c_size_t),
]
kernel32 = ctypes.WinDLL("kernel32", use_last_error=True)
kernel32.CreateJobObjectW.restype = wintypes.HANDLE
kernel32.SetInformationJobObject.argtypes = [
wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD
]
kernel32.AssignProcessToJobObject.argtypes = [wintypes.HANDLE, wintypes.HANDLE]
handle = kernel32.CreateJobObjectW(None, None)
if not handle:
raise ctypes.WinError(ctypes.get_last_error())
info = JOBOBJECT_EXTENDED_LIMIT_INFORMATION()
info.BasicLimitInformation.LimitFlags = 0x00002000 # KILL_ON_JOB_CLOSE
if not kernel32.SetInformationJobObject(handle, 9, ctypes.byref(info), ctypes.sizeof(info)):
kernel32.CloseHandle(handle)
raise ctypes.WinError(ctypes.get_last_error())
if not kernel32.AssignProcessToJobObject(handle, wintypes.HANDLE(process_handle)):
kernel32.CloseHandle(handle)
raise ctypes.WinError(ctypes.get_last_error())
self.handle = handle
def close(self) -> None:
if self.handle is None or os.name != "nt":
return
import ctypes
ctypes.WinDLL("kernel32", use_last_error=True).CloseHandle(self.handle)
self.handle = None
+26 -14
View File
@@ -118,11 +118,11 @@ PARAMETER_ENGINES = {
'chatterbox', 'chatterbox_official_23lang', 'f5tts', 'higgs_audio',
'higgs_audio_v3', 'vibevoice', 'index_tts', 'step_audio_editx', 'cosyvoice', 'qwen3_tts',
'dots_tts', 'fish_audio_s2', 'omnivoice',
'echo_tts', 'moss_tts', 'moss_soundeffect_v2', 'dramabox'
'echo_tts', 'moss_tts', 'moss_soundeffect_v2', 'dramabox', 'audio_cpp'
},
'temperature': {
'chatterbox', 'chatterbox_official_23lang', 'f5tts', 'higgs_audio',
'higgs_audio_v3', 'vibevoice', 'index_tts', 'step_audio_editx', 'qwen3_tts', 'moss_tts', 'fish_audio_s2'
'higgs_audio_v3', 'vibevoice', 'index_tts', 'step_audio_editx', 'qwen3_tts', 'moss_tts', 'fish_audio_s2', 'audio_cpp'
},
'cfg': {
'f5tts', 'vibevoice', 'index_tts', 'chatterbox', 'chatterbox_official_23lang',
@@ -150,10 +150,10 @@ PARAMETER_ENGINES = {
'dramabox'
},
'num_steps': {
'echo_tts', 'dots_tts', 'omnivoice'
'echo_tts', 'dots_tts', 'omnivoice', 'audio_cpp'
},
'guidance_scale': {
'dots_tts', 'omnivoice'
'dots_tts', 'omnivoice', 'audio_cpp'
},
'duration': {
'omnivoice'
@@ -216,13 +216,13 @@ PARAMETER_ENGINES = {
'chatterbox', 'chatterbox_official_23lang'
},
'speed': {
'f5tts', 'cosyvoice', 'omnivoice'
'f5tts', 'cosyvoice', 'omnivoice', 'audio_cpp'
},
'top_p': {
'higgs_audio', 'higgs_audio_v3', 'vibevoice', 'index_tts', 'qwen3_tts', 'moss_tts', 'fish_audio_s2'
'higgs_audio', 'higgs_audio_v3', 'vibevoice', 'index_tts', 'qwen3_tts', 'moss_tts', 'fish_audio_s2', 'audio_cpp'
},
'top_k': {
'higgs_audio', 'higgs_audio_v3', 'index_tts', 'qwen3_tts', 'moss_tts'
'higgs_audio', 'higgs_audio_v3', 'index_tts', 'qwen3_tts', 'moss_tts', 'audio_cpp'
},
'audio_temperature': {
'moss_tts'
@@ -234,7 +234,7 @@ PARAMETER_ENGINES = {
'moss_tts'
},
'repetition_penalty': {
'moss_tts', 'fish_audio_s2'
'moss_tts', 'fish_audio_s2', 'audio_cpp'
},
'audio_repetition_penalty': {
'moss_tts'
@@ -243,7 +243,7 @@ PARAMETER_ENGINES = {
'moss_tts'
},
'max_new_tokens': {
'higgs_audio_v3', 'moss_tts', 'fish_audio_s2'
'higgs_audio_v3', 'moss_tts', 'fish_audio_s2', 'audio_cpp'
},
'max_generate_length': {
'dots_tts'
@@ -252,7 +252,7 @@ PARAMETER_ENGINES = {
'moss_tts'
},
'instruction': {
'moss_tts'
'moss_tts', 'audio_cpp'
},
'quality': {
'moss_tts'
@@ -364,7 +364,10 @@ PARAMETER_NODE_KEYS = {
'ref_duration': 'ref_duration',
'rescale_scale': 'rescale_scale',
'prompt_template': 'prompt_template',
'num_steps': 'num_steps',
'num_steps': {
'default': 'num_steps',
'audio_cpp': 'num_inference_steps',
},
'guidance_scale': 'guidance_scale',
'duration': 'duration',
't_shift': 't_shift',
@@ -386,7 +389,10 @@ PARAMETER_NODE_KEYS = {
'speaker_kv_min_t': 'speaker_kv_min_t',
'sequence_length': 'sequence_length',
'exaggeration': 'exaggeration',
'speed': 'speed',
'speed': {
'default': 'speed',
'audio_cpp': 'speaking_rate',
},
'top_p': 'top_p',
'top_k': 'top_k',
'audio_temperature': 'audio_temperature',
@@ -395,10 +401,16 @@ PARAMETER_NODE_KEYS = {
'repetition_penalty': 'repetition_penalty',
'audio_repetition_penalty': 'audio_repetition_penalty',
'duration_tokens': 'duration_tokens',
'max_new_tokens': 'max_new_tokens',
'max_new_tokens': {
'default': 'max_new_tokens',
'audio_cpp': 'max_tokens',
},
'max_generate_length': 'max_generate_length',
'n_vq_for_inference': 'n_vq_for_inference',
'instruction': 'instruction',
'instruction': {
'default': 'instruction',
'audio_cpp': 'instruct',
},
'quality': 'quality',
'sound_event': 'sound_event',
'ambient_sound': 'ambient_sound',
+460
View File
@@ -0,0 +1,460 @@
import { app } from "../../scripts/app.js";
import { api } from "../../scripts/api.js";
const TARGET = "AudioCppEngineNode";
const ENDPOINT = "/api/tts-audio-suite/audio-cpp-capabilities";
const STATUS_ENDPOINT = "/api/tts-audio-suite/audio-cpp-status";
const PANEL_HEIGHT = 210;
const PANEL_MIN_WIDTH = 360;
const PANEL_BOTTOM_PADDING = 14;
const PANEL_LAYOUT_HEIGHT = PANEL_HEIGHT + PANEL_BOTTOM_PADDING;
const REQUEST_ADVANCED_WIDGETS = [
"temperature", "top_p", "top_k", "repetition_penalty",
"max_tokens", "max_steps", "num_inference_steps", "guidance_scale",
"advanced_json",
];
const OWNED_ADVANCED_WIDGETS = [
"auto_download_runtime", "auto_download_model", "show_server_console",
];
let manifestPromise;
function manifest() {
manifestPromise ??= api.fetchApi(ENDPOINT).then((response) => {
if (!response.ok) throw new Error(`Capability request failed (${response.status})`);
return response.json();
});
return manifestPromise;
}
function widget(node, name) {
return (node.widgets || []).find((item) => item.name === name);
}
function hideWidget(item) {
if (!item || item.__ttsAudioCppHidden) return;
item.__ttsAudioCppOriginalType = item.type;
item.__ttsAudioCppOriginalComputeSize = item.computeSize;
item.type = "hidden";
item.hidden = true;
item.computeSize = () => [0, -4];
if (item.element) item.element.style.display = "none";
item.__ttsAudioCppHidden = true;
}
function showWidget(item) {
if (!item || !item.__ttsAudioCppHidden) return;
item.type = item.__ttsAudioCppOriginalType;
item.computeSize = item.__ttsAudioCppOriginalComputeSize;
item.hidden = false;
if (item.element) item.element.style.display = "";
item.__ttsAudioCppHidden = false;
}
function setWidgetVisible(node, name, visible) {
const item = widget(node, name);
if (visible) showWidget(item);
else hideWidget(item);
}
function resizeNodeToContent(node) {
if (node.__ttsAudioCppResizeFrame) cancelAnimationFrame(node.__ttsAudioCppResizeFrame);
node.__ttsAudioCppResizeFrame = requestAnimationFrame(() => {
node.__ttsAudioCppResizeFrame = 0;
const computed = node.computeSize();
const width = Math.max(Number(node.size?.[0]) || 0, PANEL_MIN_WIDTH, Number(computed?.[0]) || 0);
const height = Number(computed?.[1]) || Number(node.size?.[1]) || PANEL_LAYOUT_HEIGHT;
if (Math.abs(width - node.size[0]) > 0.5 || Math.abs(height - node.size[1]) > 0.5) {
node.setSize([width, height]);
}
app.graph?.setDirtyCanvas(true, true);
});
}
function applyWidgetVisibility(node, capability) {
if (!capability) return;
const advanced = Boolean(node.__ttsAudioCppAdvancedOpen);
const mode = String(widget(node, "connection_mode")?.value || "auto");
const backend = String(widget(node, "backend")?.value || "auto");
const external = mode === "external_server";
const existingBinary = mode === "existing_binary";
const suiteTasks = new Set(capability.suite_tasks || []);
const upstreamTasks = new Set(capability.upstream_tasks || []);
setWidgetVisible(node, "package_id", !external);
setWidgetVisible(node, "task", external || upstreamTasks.size > 1 || advanced);
setWidgetVisible(node, "backend", !external);
setWidgetVisible(node, "device", !external && backend !== "cpu" && advanced);
setWidgetVisible(node, "threads", !external && (backend === "cpu" || advanced));
setWidgetVisible(node, "language", suiteTasks.has("tts") || suiteTasks.has("asr"));
setWidgetVisible(node, "server_url", external);
setWidgetVisible(node, "binary_path", existingBinary || (!external && advanced));
setWidgetVisible(node, "model_path", existingBinary || (!external && advanced));
setWidgetVisible(node, "model_id", external);
setWidgetVisible(node, "voice_id", Boolean(capability.built_in_voices));
setWidgetVisible(node, "instruct", Boolean(capability.voice_design));
for (const name of REQUEST_ADVANCED_WIDGETS) setWidgetVisible(node, name, advanced);
for (const name of OWNED_ADVANCED_WIDGETS) setWidgetVisible(node, name, !external && advanced);
const toggle = widget(node, "audio_cpp_advanced_toggle");
if (toggle) {
toggle.label = advanced ? "▾ Hide advanced settings" : "▸ Show advanced settings";
}
resizeNodeToContent(node);
}
function addAdvancedToggle(node) {
const toggle = node.addWidget("button", "▸ Show advanced settings", null, () => {
node.__ttsAudioCppAdvancedOpen = !node.__ttsAudioCppAdvancedOpen;
applyWidgetVisibility(node, node.__ttsAudioCppCapability);
});
toggle.name = "audio_cpp_advanced_toggle";
toggle.label = "▸ Show advanced settings";
toggle.options ??= {};
toggle.options.tooltip = "Show uncommon runtime paths, device tuning, sampling overrides, download controls, server debugging, and raw request JSON.";
toggle.options.serialize = false;
toggle.tooltip = toggle.options.tooltip;
toggle.serialize = false;
toggle.serializeValue = () => undefined;
return toggle;
}
function formatBytes(bytes) {
const value = Number(bytes);
if (!Number.isFinite(value) || value <= 0) return "unavailable";
if (value >= 1073741824) return `${(value / 1073741824).toFixed(2)} GB`;
return `${(value / 1048576).toFixed(1)} MB`;
}
function syncPackageChoices(node, data, capability) {
const packageWidget = widget(node, "package_id");
if (!packageWidget || !capability) return null;
const allowed = ["auto", ...(capability.packages || [])];
packageWidget.options ??= {};
packageWidget.options.values = allowed;
if (!allowed.includes(String(packageWidget.value))) packageWidget.value = "auto";
const resolvedId = packageWidget.value === "auto"
? capability.recommended_package_id
: String(packageWidget.value);
return data.packages?.[resolvedId] || null;
}
function syncTaskChoices(node, capability) {
const taskWidget = widget(node, "task");
if (!taskWidget || !capability) return;
const allowed = ["auto", ...(capability.upstream_tasks || [])];
taskWidget.options ??= {};
taskWidget.options.values = [...new Set(allowed)];
if (!taskWidget.options.values.includes(String(taskWidget.value))) taskWidget.value = "auto";
}
function normalizedUrl(value) {
return String(value || "").trim().replace(/\/$/, "");
}
function matchingStatus(node, sessions) {
const family = String(widget(node, "family")?.value || "");
const modelId = String(widget(node, "model_id")?.value || "").trim();
const mode = String(widget(node, "connection_mode")?.value || "auto");
const endpoint = normalizedUrl(widget(node, "server_url")?.value);
return (sessions || []).find((session) => {
if (modelId && session.model_id !== modelId) return false;
if (mode === "external_server" && endpoint) {
return !session.owned && normalizedUrl(session.endpoint) === endpoint;
}
return session.family === family && (mode === "external_server" ? !session.owned : true);
});
}
function renderStatus(node, panel, sessions, failed = false) {
const status = matchingStatus(node, sessions);
const light = panel.querySelector(".tts-acpp-light");
const label = panel.querySelector(".tts-acpp-status-text");
const state = failed ? "error" : (status?.state || "unchecked");
const labels = {
unchecked: "Not checked",
configured: "Server stopped",
server_ready: "Server connected · model not confirmed",
model_ready: "Server connected · model ready",
error: "Status unavailable",
};
light.dataset.state = state;
label.textContent = labels[state] || state;
const memory = panel.querySelector(".tts-acpp-memory");
const stop = panel.querySelector(".tts-acpp-stop");
node.__ttsAudioCppSessionId = status?.session_id || "";
if (status?.pid) {
const privateGb = status.private_bytes ? ` · ${(status.private_bytes / 1073741824).toFixed(1)} GB private` : "";
memory.textContent = `PID ${status.pid}${privateGb}`;
memory.hidden = false;
} else {
memory.hidden = true;
}
stop.hidden = !(status?.owned && ["server_ready", "model_ready"].includes(status.state));
}
async function refreshStatus(node, panel) {
try {
const response = await api.fetchApi(STATUS_ENDPOINT);
if (!response.ok) throw new Error(`Status request failed (${response.status})`);
const data = await response.json();
renderStatus(node, panel, data.sessions);
} catch (_error) {
renderStatus(node, panel, [], true);
}
}
function inputIsSpeaker(input) {
return /^speaker\d+$/.test(String(input?.name || ""));
}
function removeUnusedSpeakerInputs(node, maximum) {
for (let index = (node.inputs || []).length - 1; index >= 0; index -= 1) {
const input = node.inputs[index];
const number = Number(String(input?.name || "").replace("speaker", ""));
if (inputIsSpeaker(input) && input.link == null && number > maximum) node.removeInput(index);
}
}
function syncSpeakers(node, capability) {
const native = capability?.native_multi_speaker || {};
const maximum = native.supported ? Number(native.max_speakers || 1) : 1;
removeUnusedSpeakerInputs(node, maximum);
const speakers = (node.inputs || []).filter(inputIsSpeaker);
if (maximum <= 1) {
for (let index = (node.inputs || []).length - 1; index >= 0; index -= 1) {
if (inputIsSpeaker(node.inputs[index]) && node.inputs[index].link == null) node.removeInput(index);
}
return;
}
speakers.forEach((input, index) => {
input.name = `speaker${index + 2}`;
input.label = `Speaker ${index + 2}`;
});
const last = speakers[speakers.length - 1];
if (speakers.length < maximum - 1 && (!last || last.link != null)) {
node.addInput(`speaker${speakers.length + 2}`, "*");
}
}
function pill(text, tone = "normal", title = "") {
const element = document.createElement("span");
element.textContent = text;
element.className = `tts-acpp-pill ${tone}`;
if (title) element.title = title;
return element;
}
function appendAsrFeaturePills(container, capability) {
const features = capability.asr_features || {};
if (features.diarization === "native") {
container.append(pill(
"Diarization",
"special",
"Native speaker-attributed turns are available from this ASR family.",
));
} else {
container.append(pill(
"No diarization",
"muted",
"This ASR family does not return speaker identities.",
));
}
const timing = features.timing || "none";
if (timing === "native_word") {
container.append(pill(
"Word timestamps",
"info",
"Native word or token timestamps are available without a separate aligner.",
));
} else if (timing === "native_segment") {
container.append(pill(
"Segment timestamps",
"info",
"Native timed transcript or speaker segments are available.",
));
} else if (timing === "optional_forced_aligner") {
container.append(pill(
"Optional forced aligner",
"warn",
"Word timestamps require the separate Qwen3 Forced Aligner model.",
));
} else {
container.append(pill(
"No timestamps",
"muted",
"This ASR family currently returns transcription text without timing alignment.",
));
}
}
function setWidgetHeight(widget, height) {
try { widget.height = height; } catch (_error) { /* getter-only on some builds */ }
try { widget.computedHeight = height; } catch (_error) { /* getter-only on some builds */ }
}
function makePanel(node) {
const panel = document.createElement("div");
panel.className = "tts-acpp-panel";
panel.innerHTML = `<style>
.tts-acpp-panel{box-sizing:border-box;width:100%;max-width:100%;height:${PANEL_HEIGHT}px;overflow:hidden;margin:0;padding:9px 10px;border:1px solid var(--border-color,#454545);border-radius:7px;background:color-mix(in srgb,var(--comfy-menu-bg,#202020) 90%,#4d78a8 10%);color:var(--input-text,#ddd);font:12px/1.35 sans-serif}
.tts-acpp-head{display:flex;flex-wrap:wrap;justify-content:space-between;align-items:center;gap:5px 8px;margin-bottom:5px}.tts-acpp-title{font-weight:650;font-size:13px}.tts-acpp-status{display:flex;align-items:center;gap:5px;color:var(--descrip-text,#aaa);font-size:11px;white-space:nowrap}
.tts-acpp-light{width:8px;height:8px;border-radius:50%;background:#777;box-shadow:0 0 0 2px color-mix(in srgb,#777 25%,transparent)}.tts-acpp-light[data-state="configured"]{background:#d49a36}.tts-acpp-light[data-state="server_ready"]{background:#4d9bea}.tts-acpp-light[data-state="model_ready"]{background:#48bf78;box-shadow:0 0 5px #48bf78}.tts-acpp-light[data-state="error"]{background:#d85b5b}
.tts-acpp-runtime{display:flex;flex-wrap:wrap;align-items:center;justify-content:space-between;gap:4px 8px;margin:-1px 0 6px;color:var(--descrip-text,#aaa);font-size:11px}.tts-acpp-stop{border:1px solid #6d4b4b;border-radius:5px;background:#3d2929;color:#efc5c5;padding:2px 6px;cursor:pointer}.tts-acpp-stop:hover{background:#543232}
.tts-acpp-pills{display:flex;flex-wrap:wrap;gap:4px;margin-bottom:6px}
.tts-acpp-pill{padding:2px 7px;border:1px solid transparent;border-radius:999px;background:#3a4652;color:#dcecff;font-size:11px;line-height:1.35;white-space:nowrap}.tts-acpp-pill.warn{border-color:#806635;background:#5b4929;color:#ffe0a3}.tts-acpp-pill.good{border-color:#38684e;background:#294d3b;color:#bdebd2}.tts-acpp-pill.info{border-color:#365f7d;background:#29485f;color:#c8e7ff}.tts-acpp-pill.special{border-color:#685485;background:#46385d;color:#e8d9ff}.tts-acpp-pill.muted{border-color:#4d565f;background:#343b42;color:#b9c1c9}
.tts-acpp-detail{color:var(--descrip-text,#b8b8b8);margin-top:2px}.tts-acpp-summary{margin-top:6px;color:var(--input-text,#ddd)}
</style><div class="tts-acpp-head"><div class="tts-acpp-title">audio.cpp capabilities</div><div class="tts-acpp-status"><span class="tts-acpp-light" data-state="unchecked"></span><span class="tts-acpp-status-text">Not checked</span></div></div><div class="tts-acpp-runtime"><span class="tts-acpp-memory" hidden></span><button class="tts-acpp-stop" type="button" hidden>Stop owned server</button></div><div class="tts-acpp-pills"></div><div class="tts-acpp-details"></div><div class="tts-acpp-summary"></div>`;
panel.querySelector(".tts-acpp-stop").addEventListener("click", async () => {
const sessionId = node.__ttsAudioCppSessionId;
if (!sessionId) return;
const response = await api.fetchApi("/api/tts-audio-suite/audio-cpp-stop", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ session_id: sessionId }),
});
if (!response.ok) {
const data = await response.json().catch(() => ({}));
throw new Error(data.error || `Stop failed (${response.status})`);
}
refreshStatus(node, panel);
});
const panelWidget = node.addDOMWidget("audio_cpp_capabilities", "div", panel, {
serialize: false,
hideOnZoom: false,
getMinHeight: () => PANEL_LAYOUT_HEIGHT,
getHeight: () => PANEL_LAYOUT_HEIGHT,
});
panelWidget.computeSize = (inputWidth) => {
const width = Array.isArray(inputWidth) ? inputWidth[0] : inputWidth;
return [Math.max(PANEL_MIN_WIDTH, Number(width) || PANEL_MIN_WIDTH), PANEL_LAYOUT_HEIGHT];
};
panelWidget.getHeight = () => PANEL_LAYOUT_HEIGHT;
panelWidget.computeLayoutSize = () => ({
minWidth: PANEL_MIN_WIDTH,
minHeight: PANEL_LAYOUT_HEIGHT,
});
panelWidget.options ??= {};
panelWidget.options.minNodeSize = [PANEL_MIN_WIDTH, PANEL_LAYOUT_HEIGHT];
setWidgetHeight(panelWidget, PANEL_LAYOUT_HEIGHT);
if (panelWidget.element) {
panelWidget.element.style.boxSizing = "border-box";
panelWidget.element.style.width = "100%";
panelWidget.element.style.maxWidth = "100%";
panelWidget.element.style.height = `${PANEL_HEIGHT}px`;
panelWidget.element.style.minHeight = `${PANEL_HEIGHT}px`;
panelWidget.element.style.overflow = "hidden";
}
return panel;
}
function render(node, panel, data, capability) {
if (!capability) return;
node.__ttsAudioCppCapability = capability;
const selectedPackage = syncPackageChoices(node, data, capability);
syncTaskChoices(node, capability);
panel.querySelector(".tts-acpp-title").textContent = capability.display_name;
const pills = panel.querySelector(".tts-acpp-pills");
pills.replaceChildren();
const suiteTasks = new Set(capability.suite_tasks || []);
if (suiteTasks.has("tts")) pills.append(pill("TTS", "good", "Text-to-speech is wired to the Suite's Unified Text and SRT nodes."));
if (suiteTasks.has("asr")) pills.append(pill("ASR", "good", "Speech recognition is wired to the Suite's Unified ASR node."));
if (suiteTasks.has("voice_conversion")) pills.append(pill("Voice conversion", "good", "Voice conversion is wired to the Suite's Unified Voice Changer node."));
if (suiteTasks.has("asr")) appendAsrFeaturePills(pills, capability);
if (suiteTasks.has("tts") && capability.reference_audio !== "none") pills.append(pill("Voice clone", "normal", "This family accepts reference audio for voice cloning or conditioning."));
if (["required", "required_per_speaker"].includes(capability.reference_audio)) {
pills.append(pill("Reference required", "warn", "Generation requires reference audio."));
}
if (capability.reference_transcript === "required") {
pills.append(pill("Transcript required", "warn", "The transcript matching the reference audio is required."));
}
if (capability.built_in_voices) pills.append(pill("Built-in voices", "info", "This family includes model-provided voices."));
if (capability.voice_design) pills.append(pill("Voice design", "normal", "This family can synthesize from a written voice description."));
if (capability.inline_controls) pills.append(pill("Inline controls", "normal", "Suite-standard inline controls are translated for this family."));
if (capability.native_multi_speaker?.supported) {
const status = capability.native_multi_speaker.suite_status === "supported" ? "good" : "warn";
pills.append(pill(`Up to ${capability.native_multi_speaker.max_speakers} speakers`, status));
}
const transcript = capability.reference_transcript;
let taskDetails = "";
if (suiteTasks.has("tts")) {
taskDetails =
`<div class="tts-acpp-detail">Reference audio: ${capability.reference_audio.replaceAll("_", " ")}</div>` +
`<div class="tts-acpp-detail">Reference transcript: ${transcript}</div>`;
} else if (suiteTasks.has("asr")) {
const languages = capability.languages || [];
const languageSummary = languages.length > 8 ? `${languages.length} declared languages` : languages.join(", ");
taskDetails = `<div class="tts-acpp-detail">Languages: ${languageSummary || "model-defined"}</div>`;
} else if (suiteTasks.has("voice_conversion")) {
taskDetails = `<div class="tts-acpp-detail">Inputs: source audio + target reference audio</div>`;
}
panel.querySelector(".tts-acpp-details").innerHTML =
`<div class="tts-acpp-detail">Package: ${selectedPackage?.display_name || "external server / unresolved"}</div>` +
`<div class="tts-acpp-detail">Estimated download: ${formatBytes(selectedPackage?.estimated_download_bytes)}</div>` +
taskDetails +
`<div class="tts-acpp-detail">Suite: ${(capability.suite_tasks || []).join(" · ")}</div>`;
panel.querySelector(".tts-acpp-summary").textContent = capability.summary || capability.description;
syncSpeakers(node, capability);
applyWidgetVisibility(node, capability);
}
function hookWidgetCallback(node, name, callback) {
const item = widget(node, name);
if (!item || item.__ttsAudioCppCallbackHooked) return;
item.__ttsAudioCppCallbackHooked = true;
const original = item.callback;
item.callback = (...args) => {
const result = original?.apply(item, args);
callback();
return result;
};
}
app.registerExtension({
name: "TTS_Audio_Suite.AudioCppCapabilities",
async beforeRegisterNodeDef(nodeType, nodeData) {
if ((nodeData?.name || nodeData?.comfyClass) !== TARGET) return;
const original = nodeType.prototype.onNodeCreated;
nodeType.prototype.onNodeCreated = function() {
const result = original?.apply(this, arguments);
addAdvancedToggle(this);
const panel = makePanel(this);
const family = widget(this, "family");
const update = () => manifest().then((data) => render(this, panel, data, data.families?.[family?.value])).catch((error) => {
panel.querySelector(".tts-acpp-summary").textContent = error.message;
});
hookWidgetCallback(this, "family", update);
hookWidgetCallback(this, "package_id", update);
hookWidgetCallback(this, "task", update);
hookWidgetCallback(this, "connection_mode", () => applyWidgetVisibility(this, this.__ttsAudioCppCapability));
hookWidgetCallback(this, "backend", () => applyWidgetVisibility(this, this.__ttsAudioCppCapability));
const refresh = () => refreshStatus(this, panel);
this.__ttsAudioCppStatusRefresh = refresh;
setTimeout(() => { update(); refresh(); }, 0);
this.__ttsAudioCppStatusTimer = setInterval(refresh, 5000);
return result;
};
const removed = nodeType.prototype.onRemoved;
nodeType.prototype.onRemoved = function() {
clearInterval(this.__ttsAudioCppStatusTimer);
if (this.__ttsAudioCppResizeFrame) cancelAnimationFrame(this.__ttsAudioCppResizeFrame);
return removed?.apply(this, arguments);
};
const connectionChanged = nodeType.prototype.onConnectionsChange;
nodeType.prototype.onConnectionsChange = function(type, index) {
const result = connectionChanged?.apply(this, arguments);
if (inputIsSpeaker(this.inputs?.[index])) {
const family = widget(this, "family")?.value;
manifest().then((data) => syncSpeakers(this, data.families?.[family]));
}
return result;
};
},
setup() {
for (const eventName of ["executing", "executed", "execution_error"]) {
api.addEventListener(eventName, () => {
for (const node of app.graph?._nodes || []) node.__ttsAudioCppStatusRefresh?.();
});
}
},
});