Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5a28a46010 | ||
|
|
211b192f4a | ||
|
|
4a99f15851 |
@@ -355,6 +355,8 @@ def setup_api_routes():
|
||||
|
||||
from utils.voice.alias_api import register_character_alias_routes
|
||||
register_character_alias_routes(PromptServer.instance.routes, web)
|
||||
from utils.audio_cpp.capability_api import register_audio_cpp_capability_routes
|
||||
register_audio_cpp_capability_routes(PromptServer.instance.routes, web)
|
||||
|
||||
@PromptServer.instance.routes.get("/api/tts-audio-suite/index-tts-emotion-presets")
|
||||
async def get_index_tts_emotion_presets_endpoint(request):
|
||||
|
||||
@@ -0,0 +1,486 @@
|
||||
"""audio.cpp adapter for the Suite's unified ASR pipeline."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from typing import Any, Dict, Iterable, Mapping, Optional
|
||||
|
||||
import torch
|
||||
|
||||
from utils.asr.types import ASRRequest, ASRResult, ASRSegment, ASRWord
|
||||
from utils.audio.processing import AudioProcessingUtils
|
||||
|
||||
|
||||
_NATIVE_CHUNK_FAMILIES = {
|
||||
"fun_asr_nano",
|
||||
"higgs_audio_stt",
|
||||
"hviske_asr",
|
||||
"qwen3_asr",
|
||||
"vibevoice_asr",
|
||||
"voxtral_realtime",
|
||||
}
|
||||
|
||||
# audio.cpp release-0.5.1 keeps stale offline decoder state for these loaders:
|
||||
# the first request transcribes normally and later requests return empty text.
|
||||
# A fresh owned process is currently the only reliable reset contract.
|
||||
_RESTART_BETWEEN_CHUNKS_FAMILIES = {"nemotron_asr", "voxtral_realtime"}
|
||||
|
||||
|
||||
def _session(config: Mapping[str, Any]):
|
||||
from utils.audio_cpp.session import get_audio_cpp_session
|
||||
|
||||
return get_audio_cpp_session(dict(config))
|
||||
|
||||
|
||||
def _advanced_options(config: Mapping[str, Any]) -> Dict[str, Any]:
|
||||
value = config.get("advanced_options", config.get("request_options", {}))
|
||||
if value in (None, ""):
|
||||
return {}
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
value = json.loads(value)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
|
||||
if not isinstance(value, Mapping):
|
||||
raise ValueError("audio.cpp advanced options must be a JSON object")
|
||||
return dict(value)
|
||||
|
||||
|
||||
def _audio_path(audio: Mapping[str, Any]) -> str:
|
||||
waveform = audio.get("waveform")
|
||||
sample_rate = audio.get("sample_rate")
|
||||
if not torch.is_tensor(waveform):
|
||||
raise TypeError("audio.cpp ASR input must contain a waveform tensor")
|
||||
if sample_rate is None or int(sample_rate) <= 0:
|
||||
raise ValueError("audio.cpp ASR input must contain a positive sample_rate")
|
||||
return os.path.abspath(
|
||||
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
|
||||
)
|
||||
|
||||
|
||||
def _waveform_3d(audio: Mapping[str, Any]) -> tuple[torch.Tensor, int]:
|
||||
waveform = audio.get("waveform")
|
||||
sample_rate = int(audio.get("sample_rate") or 0)
|
||||
if not torch.is_tensor(waveform):
|
||||
raise TypeError("audio.cpp ASR input must contain a waveform tensor")
|
||||
if sample_rate <= 0:
|
||||
raise ValueError("audio.cpp ASR input must contain a positive sample_rate")
|
||||
if waveform.ndim == 1:
|
||||
waveform = waveform.unsqueeze(0).unsqueeze(0)
|
||||
elif waveform.ndim == 2:
|
||||
waveform = waveform.unsqueeze(0)
|
||||
elif waveform.ndim != 3:
|
||||
raise ValueError(
|
||||
"audio.cpp ASR waveform must have [samples], [channels, samples], or "
|
||||
"[batch, channels, samples] shape"
|
||||
)
|
||||
if waveform.shape[0] != 1:
|
||||
raise ValueError("audio.cpp ASR accepts one audio item at a time")
|
||||
if waveform.shape[-1] <= 0:
|
||||
raise ValueError("audio.cpp ASR input audio is empty")
|
||||
return waveform.detach().cpu(), sample_rate
|
||||
|
||||
|
||||
def _chunk_ranges(
|
||||
total_samples: int,
|
||||
sample_rate: int,
|
||||
chunk_size: int,
|
||||
overlap: int,
|
||||
) -> list[tuple[int, int]]:
|
||||
if chunk_size <= 0:
|
||||
return [(0, total_samples)]
|
||||
if overlap < 0:
|
||||
raise ValueError("ASR overlap must be zero or greater")
|
||||
if overlap >= chunk_size:
|
||||
raise ValueError("ASR overlap must be smaller than chunk_size")
|
||||
|
||||
chunk_samples = chunk_size * sample_rate
|
||||
if total_samples <= chunk_samples:
|
||||
return [(0, total_samples)]
|
||||
step_samples = (chunk_size - overlap) * sample_rate
|
||||
ranges = []
|
||||
start = 0
|
||||
while start < total_samples:
|
||||
end = min(start + chunk_samples, total_samples)
|
||||
ranges.append((start, end))
|
||||
if end >= total_samples:
|
||||
break
|
||||
start += step_samples
|
||||
return ranges
|
||||
|
||||
|
||||
def _normalized_token(value: str) -> str:
|
||||
return re.sub(r"[^\w]+", "", value, flags=re.UNICODE).casefold()
|
||||
|
||||
|
||||
def _merge_transcript(parts: Iterable[str]) -> str:
|
||||
merged: list[str] = []
|
||||
for part in parts:
|
||||
incoming = str(part or "").strip().split()
|
||||
if not incoming:
|
||||
continue
|
||||
if not merged:
|
||||
merged.extend(incoming)
|
||||
continue
|
||||
limit = min(len(merged), len(incoming), 80)
|
||||
duplicate_count = 0
|
||||
for size in range(limit, 0, -1):
|
||||
left = [_normalized_token(token) for token in merged[-size:]]
|
||||
right = [_normalized_token(token) for token in incoming[:size]]
|
||||
if all(left) and left == right:
|
||||
duplicate_count = size
|
||||
break
|
||||
merged.extend(incoming[duplicate_count:])
|
||||
return " ".join(merged).strip()
|
||||
|
||||
|
||||
def _offset_words(
|
||||
words: Iterable[ASRWord], offset: float, unique_after: Optional[float]
|
||||
) -> list[ASRWord]:
|
||||
shifted = []
|
||||
for word in words:
|
||||
item = ASRWord(start=word.start + offset, end=word.end + offset, text=word.text)
|
||||
if unique_after is not None and (item.start + item.end) / 2.0 < unique_after:
|
||||
continue
|
||||
shifted.append(item)
|
||||
return shifted
|
||||
|
||||
|
||||
def _offset_segments(
|
||||
segments: Iterable[ASRSegment], offset: float, unique_after: Optional[float]
|
||||
) -> list[ASRSegment]:
|
||||
shifted = []
|
||||
for segment in segments:
|
||||
item = ASRSegment(
|
||||
start=segment.start + offset,
|
||||
end=segment.end + offset,
|
||||
text=segment.text,
|
||||
speaker=segment.speaker,
|
||||
)
|
||||
if unique_after is not None and (item.start + item.end) / 2.0 < unique_after:
|
||||
continue
|
||||
shifted.append(item)
|
||||
return shifted
|
||||
|
||||
|
||||
def _seconds(value: Any, sample_rate: int) -> float:
|
||||
try:
|
||||
return max(0.0, float(value) / float(sample_rate))
|
||||
except (TypeError, ValueError, ZeroDivisionError):
|
||||
return 0.0
|
||||
|
||||
|
||||
def _words(payload: Mapping[str, Any], sample_rate: int) -> list[ASRWord]:
|
||||
words = []
|
||||
for item in payload.get("words") or []:
|
||||
if not isinstance(item, Mapping):
|
||||
continue
|
||||
text = str(item.get("word", item.get("text", ""))).strip()
|
||||
if not text:
|
||||
continue
|
||||
words.append(
|
||||
ASRWord(
|
||||
start=_seconds(item.get("start_sample"), sample_rate),
|
||||
end=_seconds(item.get("end_sample"), sample_rate),
|
||||
text=text,
|
||||
)
|
||||
)
|
||||
return words
|
||||
|
||||
|
||||
def _plain_segments(payload: Mapping[str, Any], sample_rate: int) -> list[ASRSegment]:
|
||||
segments = []
|
||||
for item in payload.get("segments") or []:
|
||||
if not isinstance(item, Mapping):
|
||||
continue
|
||||
text = str(item.get("text", "")).strip()
|
||||
segments.append(
|
||||
ASRSegment(
|
||||
start=_seconds(item.get("start_sample"), sample_rate),
|
||||
end=_seconds(item.get("end_sample"), sample_rate),
|
||||
text=text,
|
||||
)
|
||||
)
|
||||
return segments
|
||||
|
||||
|
||||
def _speaker_segments(payload: Mapping[str, Any], sample_rate: int) -> list[ASRSegment]:
|
||||
segments = []
|
||||
for item in payload.get("speaker_turns") or []:
|
||||
if not isinstance(item, Mapping):
|
||||
continue
|
||||
speaker = str(item.get("speaker_id", "")).strip()
|
||||
if speaker and not speaker.lower().startswith("speaker"):
|
||||
speaker = f"Speaker {speaker}"
|
||||
segments.append(
|
||||
ASRSegment(
|
||||
start=_seconds(item.get("start_sample"), sample_rate),
|
||||
end=_seconds(item.get("end_sample"), sample_rate),
|
||||
text=str(item.get("text", "")).strip(),
|
||||
speaker=speaker or None,
|
||||
)
|
||||
)
|
||||
return segments
|
||||
|
||||
|
||||
def _attach_words(segments: Iterable[ASRSegment], words: Iterable[ASRWord]) -> None:
|
||||
segment_list = list(segments)
|
||||
for word in words:
|
||||
midpoint = (word.start + word.end) / 2.0
|
||||
target = next(
|
||||
(segment for segment in segment_list if segment.start <= midpoint <= segment.end),
|
||||
None,
|
||||
)
|
||||
if target is not None:
|
||||
target.words.append(word)
|
||||
|
||||
|
||||
class AudioCppASREngineAdapter:
|
||||
"""Normalize audio.cpp transcript/timing output into ``ASRResult``."""
|
||||
|
||||
def __init__(self, engine_data: Dict[str, Any]):
|
||||
self.engine_data = dict(engine_data)
|
||||
self.config = dict(engine_data.get("config", engine_data))
|
||||
|
||||
def _session_config(self) -> Dict[str, Any]:
|
||||
config = dict(self.config)
|
||||
if str(config.get("connection_mode", "auto")).lower() != "external_server":
|
||||
config["requested_task"] = "asr"
|
||||
config["task"] = "asr"
|
||||
return config
|
||||
|
||||
def transcribe(self, req: ASRRequest) -> ASRResult:
|
||||
if req.task != "transcribe":
|
||||
raise ValueError(
|
||||
"audio.cpp release-0.5.1 ASR loaders support transcription, not the "
|
||||
"Unified ASR translate mode"
|
||||
)
|
||||
|
||||
config = self._session_config()
|
||||
family = str(config.get("family", "")).strip()
|
||||
warnings: list[str] = []
|
||||
notes: list[str] = []
|
||||
options = _advanced_options(config)
|
||||
|
||||
# VibeVoice-ASR owns diarization across its full recording. Independent
|
||||
# Suite requests can restart speaker numbering, so preserve its native
|
||||
# chunking only for this mode. All other ASR uses Suite-side windows.
|
||||
native_diarization = (
|
||||
family == "vibevoice_asr" and req.diarization and req.chunk_size > 0
|
||||
)
|
||||
if native_diarization:
|
||||
options.setdefault("audio_chunk_mode", "fixed")
|
||||
options.setdefault("audio_chunk_seconds", int(req.chunk_size))
|
||||
if req.overlap > 0:
|
||||
notes.append(
|
||||
"VibeVoice-ASR diarization uses native chunking to preserve speaker "
|
||||
"identity; the Suite overlap setting is not applied."
|
||||
)
|
||||
elif family in _NATIVE_CHUNK_FAMILIES:
|
||||
options.setdefault("audio_chunk_mode", "none")
|
||||
|
||||
if req.timestamps == "word" and family == "qwen3_asr":
|
||||
session_options = config.get("session_options") or {}
|
||||
aligner = session_options.get("qwen3_asr.forced_aligner_model_path")
|
||||
if aligner:
|
||||
options["return_timestamps"] = True
|
||||
else:
|
||||
warnings.append(
|
||||
"Qwen3-ASR word timestamps require the optional Qwen3 Forced Aligner; "
|
||||
"transcription continued without downloading that auxiliary model."
|
||||
)
|
||||
|
||||
waveform, source_rate = _waveform_3d(req.audio)
|
||||
ranges = (
|
||||
[(0, waveform.shape[-1])]
|
||||
if native_diarization
|
||||
else _chunk_ranges(
|
||||
waveform.shape[-1], source_rate, int(req.chunk_size), int(req.overlap)
|
||||
)
|
||||
)
|
||||
session = _session(config)
|
||||
if str(getattr(session, "task", "asr")) != "asr":
|
||||
raise ValueError(
|
||||
f"audio.cpp model '{session.model_id}' is configured for task "
|
||||
f"'{session.task}', not ASR"
|
||||
)
|
||||
restart_between_chunks = (
|
||||
len(ranges) > 1 and family in _RESTART_BETWEEN_CHUNKS_FAMILIES
|
||||
)
|
||||
if restart_between_chunks and not bool(getattr(session, "owned", False)):
|
||||
raise RuntimeError(
|
||||
f"audio.cpp release-0.5.1 {family} returns empty text after its first "
|
||||
"offline request. Suite-side chunking therefore requires a managed "
|
||||
"audio.cpp server so the Suite can reset it between chunks. Set "
|
||||
"connection_mode to managed, or set ASR chunk_size to 0 when using "
|
||||
"an external server."
|
||||
)
|
||||
if restart_between_chunks:
|
||||
notes.append(
|
||||
f"audio.cpp release-0.5.1 {family} requires a managed server reset "
|
||||
"between Suite chunks to avoid empty repeated-request results."
|
||||
)
|
||||
|
||||
display_family = family or "external model"
|
||||
print(f"🎧 audio.cpp ASR: Transcribing with {display_family}...")
|
||||
if len(ranges) > 1:
|
||||
notes.append(
|
||||
f"Suite-side ASR chunking used {len(ranges)} windows of "
|
||||
f"{int(req.chunk_size)}s with {int(req.overlap)}s overlap."
|
||||
)
|
||||
print(
|
||||
f"🧩 audio.cpp ASR: {len(ranges)} chunks "
|
||||
f"({int(req.chunk_size)}s, {int(req.overlap)}s overlap)"
|
||||
)
|
||||
|
||||
payloads: list[Mapping[str, Any]] = []
|
||||
chunk_timings: list[Mapping[str, Any]] = []
|
||||
chunk_diagnostics: list[Dict[str, Any]] = []
|
||||
started_at = time.time()
|
||||
for index, (start, end) in enumerate(ranges, start=1):
|
||||
if index > 1 and restart_between_chunks:
|
||||
print(
|
||||
f"🔄 audio.cpp ASR: Resetting {family} session for chunk "
|
||||
f"{index}/{len(ranges)}"
|
||||
)
|
||||
session.restart_owned_runtime()
|
||||
chunk_waveform = waveform[..., start:end]
|
||||
chunk_rms = float(torch.sqrt(torch.mean(chunk_waveform.float().square())).item())
|
||||
chunk_peak = float(chunk_waveform.float().abs().max().item())
|
||||
temp_path = _audio_path({
|
||||
"waveform": chunk_waveform,
|
||||
"sample_rate": source_rate,
|
||||
})
|
||||
try:
|
||||
request: Dict[str, Any] = {"audio": temp_path, "options": dict(options)}
|
||||
if req.language:
|
||||
request["language"] = req.language
|
||||
result = session.run(request)
|
||||
payload = result.raw if isinstance(result.raw, Mapping) else {}
|
||||
payloads.append(payload)
|
||||
if isinstance(payload.get("timing"), Mapping):
|
||||
chunk_timings.append(payload["timing"])
|
||||
chunk_diagnostics.append({
|
||||
"index": index,
|
||||
"start": round(start / source_rate, 3),
|
||||
"end": round(end / source_rate, 3),
|
||||
"rms": round(chunk_rms, 6),
|
||||
"peak": round(chunk_peak, 6),
|
||||
"text": str(payload.get("text", "")).strip(),
|
||||
"characters": len(str(payload.get("text", "")).strip()),
|
||||
"upstream_timing": (
|
||||
dict(payload["timing"])
|
||||
if isinstance(payload.get("timing"), Mapping)
|
||||
else None
|
||||
),
|
||||
})
|
||||
finally:
|
||||
try:
|
||||
os.remove(temp_path)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
if len(ranges) > 1:
|
||||
chunk_chars = len(str(payload.get("text", "")).strip())
|
||||
print(
|
||||
f" ASR chunk {index}/{len(ranges)} complete "
|
||||
f"({chunk_chars} chars, RMS {chunk_rms:.4f}, peak {chunk_peak:.4f})"
|
||||
)
|
||||
|
||||
words: list[ASRWord] = []
|
||||
speaker_segments: list[ASRSegment] = []
|
||||
plain_segments: list[ASRSegment] = []
|
||||
overlap_seconds = float(req.overlap) if len(ranges) > 1 else 0.0
|
||||
for index, ((start, _end), payload) in enumerate(zip(ranges, payloads)):
|
||||
offset = start / source_rate
|
||||
unique_after = offset + overlap_seconds if index > 0 else None
|
||||
words.extend(_offset_words(_words(payload, source_rate), offset, unique_after))
|
||||
speaker_segments.extend(
|
||||
_offset_segments(
|
||||
_speaker_segments(payload, source_rate), offset, unique_after
|
||||
)
|
||||
)
|
||||
plain_segments.extend(
|
||||
_offset_segments(_plain_segments(payload, source_rate), offset, unique_after)
|
||||
)
|
||||
|
||||
if req.diarization:
|
||||
segments = speaker_segments
|
||||
if segments:
|
||||
_attach_words(segments, words)
|
||||
else:
|
||||
warnings.append(
|
||||
f"audio.cpp {family or 'ASR model'} returned no speaker-attributed turns."
|
||||
)
|
||||
segments = plain_segments
|
||||
elif req.timestamps == "word" and words:
|
||||
segments = [
|
||||
ASRSegment(start=word.start, end=word.end, text=word.text, words=[word])
|
||||
for word in words
|
||||
]
|
||||
elif req.timestamps == "word":
|
||||
segments = plain_segments
|
||||
else:
|
||||
segments = []
|
||||
|
||||
text = _merge_transcript(payload.get("text", "") for payload in payloads)
|
||||
if req.diarization and speaker_segments:
|
||||
text = " ".join(
|
||||
f"[{segment.speaker}] {segment.text}" if segment.speaker else segment.text
|
||||
for segment in speaker_segments
|
||||
if segment.text
|
||||
).strip()
|
||||
if not text and speaker_segments:
|
||||
text = " ".join(segment.text for segment in speaker_segments if segment.text).strip()
|
||||
if req.timestamps == "word" and not words:
|
||||
warnings.append(f"audio.cpp {family or 'ASR model'} returned no word timestamps.")
|
||||
empty_chunks = sum(
|
||||
1 for payload in payloads if not str(payload.get("text", "")).strip()
|
||||
)
|
||||
if len(payloads) > 1 and empty_chunks:
|
||||
warnings.append(
|
||||
f"audio.cpp {family or 'ASR model'} returned no text for "
|
||||
f"{empty_chunks} of {len(payloads)} Suite chunks."
|
||||
)
|
||||
|
||||
raw: Dict[str, Any] = {}
|
||||
if warnings:
|
||||
raw["warnings"] = warnings
|
||||
if notes:
|
||||
raw["notes"] = notes
|
||||
if len(payloads) == 1 and chunk_timings:
|
||||
raw["timing"] = dict(chunk_timings[0])
|
||||
elif len(payloads) > 1:
|
||||
raw["timing"] = {
|
||||
"wall_ms": round((time.time() - started_at) * 1000.0, 3),
|
||||
"suite_chunks": len(payloads),
|
||||
"suite_chunk_size_seconds": int(req.chunk_size),
|
||||
"suite_overlap_seconds": int(req.overlap),
|
||||
"upstream_wall_ms": round(
|
||||
sum(float(item.get("wall_ms", 0.0)) for item in chunk_timings), 3
|
||||
),
|
||||
}
|
||||
raw["chunks"] = chunk_diagnostics
|
||||
output_language = next(
|
||||
(
|
||||
str(payload.get("language", "")).strip()
|
||||
for payload in payloads
|
||||
if str(payload.get("language", "")).strip()
|
||||
),
|
||||
str(req.language or "").strip(),
|
||||
) or None
|
||||
print(
|
||||
f"✅ audio.cpp ASR: Complete ({len(text)} chars, "
|
||||
f"{len(segments)} timed/speaker segments)"
|
||||
)
|
||||
return ASRResult(
|
||||
text=text,
|
||||
language=output_language,
|
||||
segments=segments,
|
||||
raw=raw or None,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["AudioCppASREngineAdapter"]
|
||||
@@ -0,0 +1,372 @@
|
||||
"""Adapter between the suite's TTS processors and an audio.cpp session."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
from typing import Any, Dict, Mapping, Optional, Tuple
|
||||
|
||||
import torch
|
||||
|
||||
from utils.audio.audio_hash import generate_stable_audio_component
|
||||
from utils.audio.cache import get_audio_cache
|
||||
from utils.audio.processing import AudioProcessingUtils
|
||||
from utils.voice.reference import effective_voice_audio
|
||||
|
||||
|
||||
_CACHE_SAMPLE_RATES: Dict[str, int] = {}
|
||||
_CACHE_SAMPLE_RATES_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def _get_session(config: Mapping[str, Any]):
|
||||
"""Import lazily so the node can still be discovered before optional setup."""
|
||||
from utils.audio_cpp.session import get_audio_cpp_session
|
||||
|
||||
return get_audio_cpp_session(dict(config))
|
||||
|
||||
|
||||
def _canonical_json(value: Mapping[str, Any]) -> str:
|
||||
return json.dumps(value, sort_keys=True, separators=(",", ":"), default=str)
|
||||
|
||||
|
||||
class AudioCppEngineAdapter:
|
||||
"""Build generic ``/v1/tasks/run`` requests and retain their real sample rate."""
|
||||
|
||||
_COMMON_REQUEST_FIELDS = (
|
||||
"temperature",
|
||||
"top_p",
|
||||
"top_k",
|
||||
"repetition_penalty",
|
||||
"max_tokens",
|
||||
"max_steps",
|
||||
"num_inference_steps",
|
||||
"guidance_scale",
|
||||
"speaking_rate",
|
||||
)
|
||||
|
||||
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
||||
self.config = dict(config or {})
|
||||
self.audio_cache = get_audio_cache()
|
||||
self._last_sample_rate: Optional[int] = None
|
||||
self._reference_files: Dict[str, str] = {}
|
||||
self._reference_lock = threading.RLock()
|
||||
|
||||
@property
|
||||
def sample_rate(self) -> Optional[int]:
|
||||
return self._last_sample_rate
|
||||
|
||||
def update_config(self, new_config: Optional[Dict[str, Any]]) -> None:
|
||||
self.config = dict(new_config or {})
|
||||
|
||||
@staticmethod
|
||||
def _reference_text(voice_ref: Any) -> str:
|
||||
if not isinstance(voice_ref, Mapping):
|
||||
return ""
|
||||
return str(
|
||||
voice_ref.get("reference_text")
|
||||
or voice_ref.get("prompt_text")
|
||||
or voice_ref.get("text")
|
||||
or ""
|
||||
).strip()
|
||||
|
||||
def _materialize_reference(self, voice_ref: Any) -> Tuple[Optional[str], str, str, Optional[str]]:
|
||||
"""Return path, transcript, stable hash, and the path that must be removed."""
|
||||
reference_text = AudioCppEngineAdapter._reference_text(voice_ref)
|
||||
if not isinstance(voice_ref, Mapping):
|
||||
return None, reference_text, "default_voice", None
|
||||
|
||||
audio = effective_voice_audio(voice_ref)
|
||||
if audio is None:
|
||||
return None, reference_text, "default_voice", None
|
||||
|
||||
if isinstance(audio, (str, os.PathLike)):
|
||||
path = os.path.abspath(os.path.expanduser(os.fspath(audio)))
|
||||
if not os.path.isfile(path):
|
||||
raise FileNotFoundError(f"audio.cpp reference audio not found: {path}")
|
||||
component = generate_stable_audio_component(audio_file_path=path)
|
||||
return path, reference_text, component, None
|
||||
|
||||
if isinstance(audio, Mapping):
|
||||
waveform = audio.get("waveform")
|
||||
sample_rate = audio.get("sample_rate")
|
||||
audio_dict = dict(audio)
|
||||
elif torch.is_tensor(audio):
|
||||
waveform = audio
|
||||
sample_rate = voice_ref.get("sample_rate")
|
||||
audio_dict = {"waveform": waveform, "sample_rate": sample_rate}
|
||||
else:
|
||||
raise TypeError(f"Unsupported audio.cpp voice reference type: {type(audio).__name__}")
|
||||
|
||||
if not torch.is_tensor(waveform):
|
||||
raise TypeError("audio.cpp reference audio must contain a waveform tensor")
|
||||
if sample_rate is None or int(sample_rate) <= 0:
|
||||
raise ValueError("audio.cpp reference audio must contain a positive sample_rate")
|
||||
|
||||
audio_dict["sample_rate"] = int(sample_rate)
|
||||
component = generate_stable_audio_component(reference_audio=audio_dict)
|
||||
if component not in {"ref_audio_error", "ref_audio_error_not_tensor"}:
|
||||
with self._reference_lock:
|
||||
cached_path = self._reference_files.get(component)
|
||||
if cached_path and os.path.isfile(cached_path):
|
||||
return cached_path, reference_text, component, None
|
||||
temp_path = os.path.abspath(
|
||||
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
|
||||
)
|
||||
self._reference_files[component] = temp_path
|
||||
return temp_path, reference_text, component, None
|
||||
|
||||
# Hash failures must not make unrelated references share one file.
|
||||
temp_path = os.path.abspath(
|
||||
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
|
||||
)
|
||||
return temp_path, reference_text, component, temp_path
|
||||
|
||||
def close(self) -> None:
|
||||
with self._reference_lock:
|
||||
paths = list(self._reference_files.values())
|
||||
self._reference_files.clear()
|
||||
for path in paths:
|
||||
try:
|
||||
os.remove(path)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _advanced_options(self) -> Dict[str, Any]:
|
||||
value = self.config.get(
|
||||
"advanced_options",
|
||||
self.config.get("request_options", self.config.get("advanced_json", {})),
|
||||
)
|
||||
if value in (None, ""):
|
||||
return {}
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
value = json.loads(value)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
|
||||
if not isinstance(value, Mapping):
|
||||
raise ValueError("audio.cpp advanced options must be a JSON object")
|
||||
return dict(value)
|
||||
|
||||
def _resolved_task(self, session: Any) -> str:
|
||||
requested = str(self.config.get("task", self.config.get("requested_task", "auto"))).lower()
|
||||
for source in (session, getattr(session, "config", None)):
|
||||
if source is None:
|
||||
continue
|
||||
value = source.get("task") if isinstance(source, Mapping) else getattr(source, "task", None)
|
||||
if str(value).lower() in {"tts", "clon", "vdes"}:
|
||||
return str(value).lower()
|
||||
|
||||
if requested in {"tts", "clon", "vdes"}:
|
||||
return requested
|
||||
if str(self.config.get("connection_mode", "auto")).lower() == "external_server":
|
||||
return "auto"
|
||||
try:
|
||||
from utils.audio_cpp.catalog import resolve_task
|
||||
|
||||
return str(
|
||||
resolve_task(
|
||||
self.config.get("family", ""),
|
||||
self.config.get("package_id", ""),
|
||||
requested="auto",
|
||||
)
|
||||
).lower()
|
||||
except (ImportError, KeyError, TypeError, ValueError):
|
||||
return "tts"
|
||||
|
||||
def _build_request(
|
||||
self,
|
||||
text: str,
|
||||
voice_path: Optional[str],
|
||||
reference_text: str,
|
||||
seed: int,
|
||||
advanced: Dict[str, Any],
|
||||
task: str,
|
||||
) -> Dict[str, Any]:
|
||||
request: Dict[str, Any] = {"text": text, "seed": str(int(seed)), "options": advanced}
|
||||
del task # The persistent session owns its one configured model/task.
|
||||
|
||||
language = str(self.config.get("language", "")).strip()
|
||||
if language and language.lower() not in {"auto", "none"}:
|
||||
request["language"] = language
|
||||
voice_id = str(self.config.get("voice_id", self.config.get("voice", ""))).strip()
|
||||
if voice_id:
|
||||
request["voice_id"] = voice_id
|
||||
if voice_path:
|
||||
request["voice_ref"] = voice_path
|
||||
if reference_text:
|
||||
request["reference_text"] = reference_text
|
||||
instruct = str(self.config.get("instruct", "")).strip()
|
||||
if instruct:
|
||||
request["instruct"] = instruct
|
||||
|
||||
for key in self._COMMON_REQUEST_FIELDS:
|
||||
value = self.config.get(key)
|
||||
if value is not None and value != "":
|
||||
request[key] = value
|
||||
return request
|
||||
|
||||
def _cache_key(
|
||||
self,
|
||||
text: str,
|
||||
audio_component: str,
|
||||
reference_text: str,
|
||||
seed: int,
|
||||
task: str,
|
||||
advanced: Dict[str, Any],
|
||||
character_name: Optional[str],
|
||||
session: Any,
|
||||
) -> str:
|
||||
session_config = getattr(session, "config", {})
|
||||
if not isinstance(session_config, Mapping):
|
||||
session_config = {}
|
||||
session_family = getattr(session, "family", None) or session_config.get(
|
||||
"family", self.config.get("family", "")
|
||||
)
|
||||
session_model_id = getattr(session, "model_id", None) or session_config.get(
|
||||
"model_id", self.config.get("model_id", "")
|
||||
)
|
||||
# Owned servers use a random loopback port on every restart; that port is
|
||||
# transport state, not model identity. External endpoints are stable and
|
||||
# must participate in the cache key.
|
||||
if bool(getattr(session, "owned", False)):
|
||||
session_endpoint = ""
|
||||
else:
|
||||
session_endpoint = getattr(session, "endpoint", None) or self.config.get(
|
||||
"server_url", self.config.get("external_server_url", "")
|
||||
)
|
||||
extra_identity = {
|
||||
"options": advanced,
|
||||
"speaking_rate": self.config.get("speaking_rate"),
|
||||
"connection_mode": self.config.get("connection_mode", "auto"),
|
||||
"server_url": session_endpoint,
|
||||
"binary_path": session_config.get("binary_path", self.config.get("binary_path", "")),
|
||||
"backend": session_config.get("backend", self.config.get("backend", "")),
|
||||
"device": session_config.get("device", self.config.get("device", "")),
|
||||
"load_options": session_config.get("load_options", self.config.get("load_options", {})),
|
||||
"session_options": session_config.get(
|
||||
"session_options", self.config.get("session_options", {})
|
||||
),
|
||||
"default_request_options": session_config.get(
|
||||
"default_request_options", self.config.get("default_request_options", {})
|
||||
),
|
||||
}
|
||||
return self.audio_cache.generate_cache_key(
|
||||
"audio_cpp",
|
||||
text=text,
|
||||
audio_component=audio_component,
|
||||
reference_text=reference_text,
|
||||
family=session_family,
|
||||
package_id=session_config.get("package_id", self.config.get("package_id", "")),
|
||||
model_path=session_config.get("model_path", self.config.get("model_path", "")),
|
||||
model_id=session_model_id,
|
||||
task=task,
|
||||
language=self.config.get("language", ""),
|
||||
voice_id=self.config.get("voice_id", self.config.get("voice", "")),
|
||||
instruct=self.config.get("instruct", ""),
|
||||
temperature=self.config.get("temperature"),
|
||||
top_p=self.config.get("top_p"),
|
||||
top_k=self.config.get("top_k"),
|
||||
repetition_penalty=self.config.get("repetition_penalty"),
|
||||
max_tokens=self.config.get("max_tokens"),
|
||||
max_steps=self.config.get("max_steps"),
|
||||
num_inference_steps=self.config.get("num_inference_steps"),
|
||||
guidance_scale=self.config.get("guidance_scale"),
|
||||
seed=int(seed),
|
||||
request_options=_canonical_json(extra_identity),
|
||||
character=character_name or "narrator",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_result(result: Any) -> Tuple[torch.Tensor, int]:
|
||||
waveform = result.get("waveform") if isinstance(result, Mapping) else getattr(result, "waveform", None)
|
||||
sample_rate = result.get("sample_rate") if isinstance(result, Mapping) else getattr(result, "sample_rate", None)
|
||||
|
||||
if waveform is None:
|
||||
named = result.get("named_audio", {}) if isinstance(result, Mapping) else getattr(result, "named_audio", {})
|
||||
values = list(named.values()) if isinstance(named, Mapping) else list(named or [])
|
||||
if len(values) == 1:
|
||||
item = values[0]
|
||||
waveform = item.get("waveform") if isinstance(item, Mapping) else getattr(item, "waveform", None)
|
||||
sample_rate = sample_rate or (item.get("sample_rate") if isinstance(item, Mapping) else getattr(item, "sample_rate", None))
|
||||
|
||||
if waveform is None:
|
||||
raise RuntimeError("audio.cpp returned no primary audio output")
|
||||
if not torch.is_tensor(waveform):
|
||||
waveform = torch.as_tensor(waveform, dtype=torch.float32)
|
||||
waveform = waveform.detach().to(device="cpu", dtype=torch.float32)
|
||||
if waveform.dim() == 1:
|
||||
waveform = waveform.unsqueeze(0)
|
||||
elif waveform.dim() == 3 and waveform.shape[0] == 1:
|
||||
waveform = waveform.squeeze(0)
|
||||
if waveform.dim() != 2:
|
||||
raise ValueError(f"audio.cpp waveform must be [channels, samples], got {tuple(waveform.shape)}")
|
||||
if sample_rate is None or int(sample_rate) <= 0:
|
||||
raise ValueError("audio.cpp returned an invalid sample rate")
|
||||
return waveform.contiguous(), int(sample_rate)
|
||||
|
||||
def generate_single(
|
||||
self,
|
||||
text: str,
|
||||
voice_ref: Optional[Dict[str, Any]] = None,
|
||||
seed: int = 0,
|
||||
enable_audio_cache: bool = True,
|
||||
character_name: Optional[str] = None,
|
||||
) -> Tuple[torch.Tensor, int]:
|
||||
stripped = str(text or "").strip()
|
||||
if not stripped:
|
||||
if self._last_sample_rate is None:
|
||||
raise ValueError("audio.cpp cannot determine a sample rate for empty text")
|
||||
return torch.zeros(1, 0, dtype=torch.float32), self._last_sample_rate
|
||||
|
||||
session = _get_session(self.config)
|
||||
task = self._resolved_task(session)
|
||||
advanced = self._advanced_options()
|
||||
cleanup_path: Optional[str] = None
|
||||
try:
|
||||
voice_path, reference_text, audio_component, cleanup_path = self._materialize_reference(voice_ref)
|
||||
cache_key = self._cache_key(
|
||||
stripped,
|
||||
audio_component,
|
||||
reference_text,
|
||||
seed,
|
||||
task,
|
||||
advanced,
|
||||
character_name,
|
||||
session,
|
||||
)
|
||||
if enable_audio_cache:
|
||||
cached = self.audio_cache.get_cached_audio(cache_key)
|
||||
with _CACHE_SAMPLE_RATES_LOCK:
|
||||
cached_rate = _CACHE_SAMPLE_RATES.get(cache_key)
|
||||
if cached is not None and cached_rate is not None:
|
||||
self._last_sample_rate = cached_rate
|
||||
return cached[0].clone(), cached_rate
|
||||
|
||||
request = self._build_request(stripped, voice_path, reference_text, seed, advanced, task)
|
||||
waveform, sample_rate = self._normalize_result(session.run(request))
|
||||
self._last_sample_rate = sample_rate
|
||||
if enable_audio_cache:
|
||||
duration = waveform.shape[-1] / sample_rate
|
||||
self.audio_cache.cache_audio(cache_key, waveform, duration)
|
||||
with _CACHE_SAMPLE_RATES_LOCK:
|
||||
_CACHE_SAMPLE_RATES[cache_key] = sample_rate
|
||||
return waveform, sample_rate
|
||||
finally:
|
||||
if cleanup_path:
|
||||
try:
|
||||
os.remove(cleanup_path)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
|
||||
|
||||
# Short alias for callers that do not use the older ``EngineAdapter`` suffix.
|
||||
AudioCppAdapter = AudioCppEngineAdapter
|
||||
@@ -0,0 +1,111 @@
|
||||
"""audio.cpp adapter for the Suite's unified Voice Changer node."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict, Mapping
|
||||
|
||||
import torch
|
||||
|
||||
from engines.adapters.audio_cpp_adapter import AudioCppEngineAdapter
|
||||
from utils.audio.processing import AudioProcessingUtils
|
||||
|
||||
|
||||
def _advanced_options(config: Mapping[str, Any]) -> Dict[str, Any]:
|
||||
value = config.get("advanced_options", config.get("request_options", {}))
|
||||
if value in (None, ""):
|
||||
return {}
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
value = json.loads(value)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
|
||||
if not isinstance(value, Mapping):
|
||||
raise ValueError("audio.cpp advanced options must be a JSON object")
|
||||
return dict(value)
|
||||
|
||||
|
||||
def _materialize(audio: Mapping[str, Any], label: str) -> str:
|
||||
waveform = audio.get("waveform")
|
||||
sample_rate = audio.get("sample_rate")
|
||||
if not torch.is_tensor(waveform):
|
||||
raise TypeError(f"audio.cpp {label} must contain a waveform tensor")
|
||||
if sample_rate is None or int(sample_rate) <= 0:
|
||||
raise ValueError(f"audio.cpp {label} must contain a positive sample_rate")
|
||||
return os.path.abspath(
|
||||
AudioProcessingUtils.save_audio_to_temp_file(waveform, int(sample_rate))
|
||||
)
|
||||
|
||||
|
||||
class AudioCppVoiceConversionAdapter:
|
||||
"""Convert source audio toward a target reference using an audio.cpp VC task."""
|
||||
|
||||
def __init__(self, config: Dict[str, Any]):
|
||||
self.config = dict(config)
|
||||
|
||||
def _session_config(self) -> Dict[str, Any]:
|
||||
config = dict(self.config)
|
||||
if str(config.get("connection_mode", "auto")).lower() != "external_server":
|
||||
config["requested_task"] = "vc"
|
||||
config["task"] = "vc"
|
||||
return config
|
||||
|
||||
def convert_voice(
|
||||
self,
|
||||
source_audio: Dict[str, Any],
|
||||
target_audio: Dict[str, Any],
|
||||
refinement_passes: int = 1,
|
||||
) -> tuple[Dict[str, Any], str]:
|
||||
from utils.audio_cpp.session import get_audio_cpp_session
|
||||
|
||||
config = self._session_config()
|
||||
family = str(config.get("family", "")).strip()
|
||||
passes = max(1, int(refinement_passes))
|
||||
current = source_audio
|
||||
output_rate = int(source_audio["sample_rate"])
|
||||
|
||||
session = get_audio_cpp_session(config)
|
||||
if str(getattr(session, "task", "vc")) != "vc":
|
||||
raise ValueError(
|
||||
f"audio.cpp model '{session.model_id}' is configured for task "
|
||||
f"'{session.task}', not voice conversion"
|
||||
)
|
||||
|
||||
for pass_index in range(passes):
|
||||
source_path = _materialize(current, "source audio")
|
||||
target_path = _materialize(target_audio, "target reference audio")
|
||||
try:
|
||||
request = {
|
||||
"audio": source_path,
|
||||
"voice_ref": target_path,
|
||||
"source_audio": source_path,
|
||||
"target_voice": target_path,
|
||||
"options": _advanced_options(config),
|
||||
}
|
||||
print(
|
||||
f"🔄 audio.cpp VC: {family or 'external model'} pass "
|
||||
f"{pass_index + 1}/{passes}..."
|
||||
)
|
||||
result = session.run(request)
|
||||
waveform, output_rate = AudioCppEngineAdapter._normalize_result(result)
|
||||
current = {"waveform": waveform.unsqueeze(0), "sample_rate": output_rate}
|
||||
finally:
|
||||
for path in (source_path, target_path):
|
||||
try:
|
||||
os.remove(path)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
|
||||
info = (
|
||||
f"Model family: {family or getattr(session, 'family', 'external')}\n"
|
||||
f"Model ID: {session.model_id}\n"
|
||||
f"Task: voice conversion\n"
|
||||
f"Refinement passes: {passes}\n"
|
||||
f"Output sample rate: {output_rate} Hz\n"
|
||||
"Conversion completed successfully"
|
||||
)
|
||||
return current, info
|
||||
|
||||
|
||||
__all__ = ["AudioCppVoiceConversionAdapter"]
|
||||
@@ -195,6 +195,14 @@ except Exception as e:
|
||||
print(f"❌ Fish Audio S2 Pro Engine failed: {e}")
|
||||
FISH_AUDIO_S2_ENGINE_AVAILABLE = False
|
||||
|
||||
try:
|
||||
audio_cpp_engine_module = load_node_module("audio_cpp_engine_node", "engines/audio_cpp_engine_node.py")
|
||||
AudioCppEngineNode = audio_cpp_engine_module.AudioCppEngineNode
|
||||
AUDIO_CPP_ENGINE_AVAILABLE = True
|
||||
except Exception as e:
|
||||
print(f"❌ audio.cpp Engine failed: {e}")
|
||||
AUDIO_CPP_ENGINE_AVAILABLE = False
|
||||
|
||||
try:
|
||||
omnivoice_engine_module = load_node_module("omnivoice_engine_node", "engines/omnivoice_engine_node.py")
|
||||
OmniVoiceEngineNode = omnivoice_engine_module.OmniVoiceEngineNode
|
||||
@@ -702,6 +710,10 @@ if FISH_AUDIO_S2_ENGINE_AVAILABLE:
|
||||
NODE_CLASS_MAPPINGS["FishAudioS2EngineNode"] = FishAudioS2EngineNode
|
||||
NODE_DISPLAY_NAME_MAPPINGS["FishAudioS2EngineNode"] = "⚙️ Fish Audio S2 Pro Engine"
|
||||
|
||||
if AUDIO_CPP_ENGINE_AVAILABLE:
|
||||
NODE_CLASS_MAPPINGS["AudioCppEngineNode"] = AudioCppEngineNode
|
||||
NODE_DISPLAY_NAME_MAPPINGS["AudioCppEngineNode"] = "⚙️ audio.cpp Multi-TTS Engine"
|
||||
|
||||
if OMNIVOICE_ENGINE_AVAILABLE:
|
||||
NODE_CLASS_MAPPINGS["OmniVoiceEngineNode"] = OmniVoiceEngineNode
|
||||
NODE_DISPLAY_NAME_MAPPINGS["OmniVoiceEngineNode"] = "⚙️ OmniVoice Engine"
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
"""audio.cpp processor exports."""
|
||||
|
||||
from .audio_cpp_processor import AudioCPPProcessor, AudioCppProcessor
|
||||
from .audio_cpp_srt_processor import (
|
||||
AudioCPPSRTProcessor,
|
||||
AudioCppSRTProcessor,
|
||||
AudioCppSubtitleProcessor,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"AudioCppProcessor",
|
||||
"AudioCPPProcessor",
|
||||
"AudioCppSRTProcessor",
|
||||
"AudioCPPSRTProcessor",
|
||||
"AudioCppSubtitleProcessor",
|
||||
]
|
||||
@@ -0,0 +1,412 @@
|
||||
"""Text orchestration for the generic audio.cpp TTS engine."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Any, Dict, List, Mapping, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
|
||||
from utils.audio.chunk_combiner import ChunkCombiner
|
||||
from utils.text.character_parser import character_parser
|
||||
from utils.text.pause_processor import PauseTagProcessor
|
||||
from utils.text.segment_parameters import ParameterValidator, apply_segment_parameters
|
||||
from utils.text.step_audio_editx_special_tags import get_edit_tags_for_segment
|
||||
from utils.voice.character_logging import (
|
||||
format_resolved_character_block,
|
||||
resolved_character_label,
|
||||
)
|
||||
from utils.voice.discovery import get_available_characters, get_character_mapping, voice_discovery
|
||||
from utils.voice.reference import effective_voice_audio
|
||||
|
||||
|
||||
class AudioCppProcessor:
|
||||
"""Apply suite text features while accepting the runtime's response sample rate."""
|
||||
|
||||
_RUNTIME_KEYS = (
|
||||
"connection_mode",
|
||||
"server_url",
|
||||
"external_server_url",
|
||||
"binary_path",
|
||||
"family",
|
||||
"package_id",
|
||||
"model_path",
|
||||
"model_id",
|
||||
"task",
|
||||
"backend",
|
||||
"device",
|
||||
)
|
||||
|
||||
def __init__(self, adapter: Any, engine_config: Optional[Dict[str, Any]] = None):
|
||||
self.adapter = adapter
|
||||
self.config = dict(engine_config or {})
|
||||
self._sample_rate: Optional[int] = None
|
||||
|
||||
@property
|
||||
def sample_rate(self) -> Optional[int]:
|
||||
return self._sample_rate
|
||||
|
||||
def update_config(self, new_config: Optional[Dict[str, Any]]) -> None:
|
||||
new_value = dict(new_config or {})
|
||||
old_signature = tuple(self.config.get(key) for key in self._RUNTIME_KEYS)
|
||||
new_signature = tuple(new_value.get(key) for key in self._RUNTIME_KEYS)
|
||||
if old_signature != new_signature:
|
||||
self._sample_rate = None
|
||||
self.config = new_value
|
||||
self.adapter.update_config(new_value)
|
||||
|
||||
def reset_sample_rate(self) -> None:
|
||||
"""Begin a top-level generation without retaining an old server rate."""
|
||||
self._sample_rate = None
|
||||
|
||||
@staticmethod
|
||||
def _check_interrupt() -> None:
|
||||
try:
|
||||
import comfy.model_management as model_management
|
||||
|
||||
if getattr(model_management, "interrupt_processing", False) is True:
|
||||
raise InterruptedError("audio.cpp generation interrupted by user")
|
||||
except ImportError:
|
||||
return
|
||||
|
||||
def _adopt_sample_rate(self, sample_rate: Any) -> int:
|
||||
try:
|
||||
value = int(sample_rate)
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ValueError(f"audio.cpp returned invalid sample rate: {sample_rate!r}") from exc
|
||||
if value <= 0:
|
||||
raise ValueError(f"audio.cpp returned invalid sample rate: {value}")
|
||||
if self._sample_rate is None:
|
||||
self._sample_rate = value
|
||||
elif self._sample_rate != value:
|
||||
raise RuntimeError(
|
||||
"audio.cpp returned inconsistent sample rates in one generation "
|
||||
f"({self._sample_rate} Hz then {value} Hz)"
|
||||
)
|
||||
return value
|
||||
|
||||
def _setup_character_parser(self, text: str) -> None:
|
||||
language = str(self.config.get("language", "auto") or "auto").strip()
|
||||
fallback = "en" if language.lower() in {"", "auto", "none"} else language.lower()
|
||||
character_parser.language_resolver.default_language = fallback
|
||||
character_parser.default_language = fallback
|
||||
|
||||
tagged = []
|
||||
for raw in re.findall(r"\[([^\]]+)\]", text or ""):
|
||||
name = raw.split("|", 1)[0].strip()
|
||||
if name and not name.lower().startswith(("pause:", "wait:", "stop:")):
|
||||
tagged.append(name)
|
||||
|
||||
available = {str(item).lower() for item in (get_available_characters() or [])}
|
||||
for alias, target in voice_discovery.get_character_aliases().items():
|
||||
available.update((str(alias).lower(), str(target).lower()))
|
||||
available.update(name.lower() for name in tagged)
|
||||
available.add("narrator")
|
||||
character_parser.set_available_characters(sorted(available))
|
||||
for character, default_language in voice_discovery.get_character_language_defaults().items():
|
||||
character_parser.set_character_language_default(character, default_language)
|
||||
character_parser.reset_session_cache()
|
||||
|
||||
@staticmethod
|
||||
def _should_apply_segment_language(segment: Any, base_config: Mapping[str, Any]) -> bool:
|
||||
language = str(getattr(segment, "language", "") or "").strip()
|
||||
if not language:
|
||||
return False
|
||||
if getattr(segment, "explicit_language", False):
|
||||
return True
|
||||
global_language = str(base_config.get("language", "auto") or "auto").strip().lower()
|
||||
parser_fallback = str(character_parser.default_language or "").strip().lower()
|
||||
return language.lower() != parser_fallback and language.lower() != global_language
|
||||
|
||||
@staticmethod
|
||||
def _voice_for_character(
|
||||
character: str,
|
||||
voice_mapping: Mapping[str, Any],
|
||||
discovered: Mapping[str, Tuple[Optional[str], Optional[str]]],
|
||||
) -> Dict[str, Any]:
|
||||
narrator = voice_mapping.get("narrator", {})
|
||||
voice = dict(narrator) if isinstance(narrator, Mapping) else {"audio": narrator}
|
||||
if character != "narrator" and character in voice_mapping:
|
||||
selected = voice_mapping[character]
|
||||
return dict(selected) if isinstance(selected, Mapping) else {"audio": selected}
|
||||
if character != "narrator":
|
||||
audio_path, reference_text = discovered.get(character, (None, None))
|
||||
if audio_path:
|
||||
return {"audio_path": audio_path, "reference_text": reference_text or ""}
|
||||
return voice
|
||||
|
||||
@staticmethod
|
||||
def _chunks(text: str, enabled: bool, max_chars: int) -> List[str]:
|
||||
if not enabled:
|
||||
return [text]
|
||||
from utils.text.chunking import ImprovedChatterBoxChunker
|
||||
|
||||
limit = ImprovedChatterBoxChunker.validate_chunking_params(max_chars)
|
||||
return ImprovedChatterBoxChunker.split_into_chunks(text, max_chars=limit)
|
||||
|
||||
@staticmethod
|
||||
def _voice_log_note(voice_ref: Mapping[str, Any]) -> str:
|
||||
if not isinstance(voice_ref, Mapping) or effective_voice_audio(voice_ref) is None:
|
||||
return " [no voice reference - model default]"
|
||||
reference_text = str(voice_ref.get("reference_text") or "").strip()
|
||||
if reference_text:
|
||||
return f" [ref text: {len(reference_text)} chars]"
|
||||
return ""
|
||||
|
||||
@staticmethod
|
||||
def _format_parameter_log(
|
||||
parameters: Mapping[str, Any], current_config: Mapping[str, Any], current_seed: int
|
||||
) -> str:
|
||||
if not parameters:
|
||||
return ""
|
||||
parts = []
|
||||
for key in parameters:
|
||||
if key == "seed":
|
||||
value = current_seed
|
||||
else:
|
||||
value = current_config.get(key, parameters.get(key))
|
||||
if value is not None and value != "":
|
||||
parts.append(f"{key}={value}")
|
||||
return ", ".join(parts)
|
||||
|
||||
def _log_generation_text(
|
||||
self,
|
||||
character: str,
|
||||
text: str,
|
||||
voice_ref: Mapping[str, Any],
|
||||
language: str,
|
||||
family: str,
|
||||
chunk_count: int,
|
||||
parameter_log: str,
|
||||
) -> None:
|
||||
display_name = resolved_character_label(character, voice_ref)
|
||||
voice_note = self._voice_log_note(voice_ref)
|
||||
print(
|
||||
f"🎭 Audio.cpp ({family}) - Generating for '{display_name}' "
|
||||
f"(Language: {language}){voice_note}:"
|
||||
)
|
||||
if parameter_log:
|
||||
print(f"🎛️ Audio.cpp params: {parameter_log}")
|
||||
print(format_resolved_character_block(character, text, voice_ref))
|
||||
if chunk_count > 1:
|
||||
print(
|
||||
f"📝 Chunking {display_name}'s text into {chunk_count} chunks "
|
||||
f"(Language: {language}){voice_note}"
|
||||
)
|
||||
|
||||
def get_character_order(self, text: str) -> List[str]:
|
||||
self._setup_character_parser(text)
|
||||
seen: List[str] = []
|
||||
for segment in character_parser.parse_text_segments(text, engine_type="audio_cpp"):
|
||||
character = segment.character or "narrator"
|
||||
if character not in seen:
|
||||
seen.append(character)
|
||||
return seen
|
||||
|
||||
def process_text(
|
||||
self,
|
||||
text: str,
|
||||
voice_mapping: Optional[Dict[str, Any]],
|
||||
seed: int,
|
||||
enable_chunking: bool = True,
|
||||
max_chars_per_chunk: int = 400,
|
||||
chunk_combination_method: str = "auto",
|
||||
silence_between_chunks_ms: int = 100,
|
||||
enable_audio_cache: bool = True,
|
||||
apply_edit_postprocessing: bool = True,
|
||||
show_text_logging: bool = True,
|
||||
reset_sample_rate: bool = True,
|
||||
**_: Any,
|
||||
) -> List[Dict[str, Any]]:
|
||||
del chunk_combination_method, silence_between_chunks_ms
|
||||
if reset_sample_rate:
|
||||
self.reset_sample_rate()
|
||||
self._check_interrupt()
|
||||
voice_mapping = dict(voice_mapping or {})
|
||||
self._setup_character_parser(text)
|
||||
base_config = self.config.copy()
|
||||
segments = character_parser.parse_text_segments(text, engine_type="audio_cpp")
|
||||
if not segments and str(text or "").strip():
|
||||
segments = character_parser.parse_text_segments(
|
||||
f"[narrator]{text}", engine_type="audio_cpp"
|
||||
)
|
||||
|
||||
characters = list({segment.character for segment in segments if segment.character})
|
||||
# GLM-TTS requires the transcript paired with its reference voice.
|
||||
# Other pinned families accept audio-only discovery and still receive a
|
||||
# transcript whenever one exists beside the character audio file.
|
||||
try:
|
||||
from utils.audio_cpp.capabilities import get_capability
|
||||
|
||||
transcript_requirement = get_capability(
|
||||
str(base_config.get("family", ""))
|
||||
)["reference_transcript"]
|
||||
except (ImportError, KeyError, ValueError):
|
||||
transcript_requirement = "none"
|
||||
discovery_type = (
|
||||
"audio_and_text" if transcript_requirement == "required" else "audio_only"
|
||||
)
|
||||
discovered = get_character_mapping(characters, engine_type=discovery_type)
|
||||
configured_speakers = list(base_config.get("speaker_references") or [])
|
||||
ordered_characters = []
|
||||
for segment in segments:
|
||||
name = segment.character or "narrator"
|
||||
if name not in ordered_characters:
|
||||
ordered_characters.append(name)
|
||||
for index, reference in enumerate(configured_speakers, start=1):
|
||||
if index < len(ordered_characters):
|
||||
selected = reference if isinstance(reference, Mapping) else {"audio": reference}
|
||||
voice_mapping[ordered_characters[index]] = dict(selected)
|
||||
records: List[Dict[str, Any]] = []
|
||||
|
||||
for segment in segments:
|
||||
self._check_interrupt()
|
||||
segment_text = str(segment.text or "").strip()
|
||||
if not segment_text:
|
||||
continue
|
||||
character = segment.character or "narrator"
|
||||
parameters = dict(segment.parameters or {})
|
||||
filtered_parameters: Dict[str, Any] = {}
|
||||
current_config = base_config
|
||||
current_seed = int(seed)
|
||||
if parameters:
|
||||
filtered_parameters = ParameterValidator.filter_parameters_for_engine(
|
||||
parameters, "audio_cpp"
|
||||
)
|
||||
current_config = apply_segment_parameters(base_config, parameters, "audio_cpp")
|
||||
current_seed = int(current_config.get("seed", seed))
|
||||
if self._should_apply_segment_language(segment, base_config):
|
||||
current_config = current_config.copy()
|
||||
current_config["language"] = segment.language
|
||||
self.adapter.update_config(current_config)
|
||||
voice_ref = self._voice_for_character(character, voice_mapping, discovered)
|
||||
try:
|
||||
from utils.audio_cpp.capabilities import CapabilityError, validate_voice_reference
|
||||
except ImportError:
|
||||
validate_voice_reference = None
|
||||
if validate_voice_reference is not None:
|
||||
try:
|
||||
validate_voice_reference(
|
||||
str(base_config.get("family", "")), voice_ref, character
|
||||
)
|
||||
except CapabilityError:
|
||||
# Preserve lightweight processor use before a concrete family
|
||||
# has been selected, while enforcing every known family.
|
||||
pass
|
||||
|
||||
def generate_fragment(content: str, edit_tags: List[Any]) -> None:
|
||||
chunks = self._chunks(content, enable_chunking, max_chars_per_chunk)
|
||||
if show_text_logging:
|
||||
language = str(current_config.get("language", "auto") or "auto")
|
||||
family = str(current_config.get("family", "unknown") or "unknown")
|
||||
self._log_generation_text(
|
||||
character,
|
||||
content,
|
||||
voice_ref,
|
||||
language,
|
||||
family,
|
||||
len(chunks),
|
||||
self._format_parameter_log(
|
||||
filtered_parameters, current_config, current_seed
|
||||
),
|
||||
)
|
||||
for chunk_index, chunk in enumerate(chunks):
|
||||
self._check_interrupt()
|
||||
waveform, response_rate = self.adapter.generate_single(
|
||||
text=chunk,
|
||||
voice_ref=voice_ref,
|
||||
seed=current_seed + chunk_index,
|
||||
enable_audio_cache=enable_audio_cache,
|
||||
character_name=character,
|
||||
)
|
||||
sample_rate = self._adopt_sample_rate(response_rate)
|
||||
waveform = waveform.detach().to(device="cpu", dtype=torch.float32)
|
||||
if waveform.dim() == 1:
|
||||
waveform = waveform.unsqueeze(0)
|
||||
if waveform.dim() != 2:
|
||||
raise ValueError(
|
||||
f"audio.cpp waveform must be [channels, samples], got {tuple(waveform.shape)}"
|
||||
)
|
||||
records.append(
|
||||
{
|
||||
"waveform": waveform,
|
||||
"sample_rate": sample_rate,
|
||||
"text": chunk,
|
||||
"edit_tags": edit_tags if chunk_index == 0 else [],
|
||||
}
|
||||
)
|
||||
|
||||
if PauseTagProcessor.has_pause_tags(segment_text):
|
||||
pause_parts, _ = PauseTagProcessor.parse_pause_tags(segment_text)
|
||||
for part_type, content in pause_parts:
|
||||
if part_type == "text":
|
||||
clean_text, edit_tags = get_edit_tags_for_segment(str(content))
|
||||
if clean_text.strip():
|
||||
generate_fragment(clean_text.strip(), edit_tags)
|
||||
else:
|
||||
records.append(
|
||||
{
|
||||
"pause_duration": float(content),
|
||||
"text": f"[pause:{content}s]",
|
||||
"edit_tags": [],
|
||||
}
|
||||
)
|
||||
else:
|
||||
clean_text, edit_tags = get_edit_tags_for_segment(segment_text)
|
||||
if clean_text.strip():
|
||||
generate_fragment(clean_text.strip(), edit_tags)
|
||||
|
||||
self.adapter.update_config(base_config)
|
||||
if any("pause_duration" in record for record in records):
|
||||
if self._sample_rate is None:
|
||||
raise ValueError("audio.cpp cannot render pauses before any response sample rate is known")
|
||||
for record in records:
|
||||
if "pause_duration" not in record:
|
||||
continue
|
||||
record["waveform"] = PauseTagProcessor.create_silence_segment(
|
||||
record.pop("pause_duration"), self._sample_rate, torch.device("cpu"), torch.float32
|
||||
)
|
||||
record["sample_rate"] = self._sample_rate
|
||||
|
||||
if apply_edit_postprocessing and records and any(record.get("edit_tags") for record in records):
|
||||
from utils.audio.edit_post_processor import process_segments as apply_edits
|
||||
|
||||
records = apply_edits(records, engine_config=base_config)
|
||||
for record in records:
|
||||
self._adopt_sample_rate(record.get("sample_rate"))
|
||||
return records
|
||||
|
||||
def combine_audio_segments(
|
||||
self,
|
||||
segments: List[Dict[str, Any]],
|
||||
method: str = "auto",
|
||||
silence_ms: int = 100,
|
||||
original_text: str = "",
|
||||
return_info: bool = False,
|
||||
) -> Union[torch.Tensor, Tuple[torch.Tensor, Dict[str, Any]]]:
|
||||
if not segments:
|
||||
empty = torch.zeros(0, dtype=torch.float32)
|
||||
return (empty, {}) if return_info else empty
|
||||
|
||||
rates = {self._adopt_sample_rate(segment.get("sample_rate")) for segment in segments}
|
||||
if len(rates) != 1:
|
||||
raise RuntimeError(f"audio.cpp segments use inconsistent sample rates: {sorted(rates)}")
|
||||
sample_rate = rates.pop()
|
||||
waveforms = [segment["waveform"] for segment in segments]
|
||||
text_chunks = [str(segment.get("text", "")) for segment in segments]
|
||||
result = ChunkCombiner.combine_chunks(
|
||||
audio_segments=waveforms,
|
||||
method=method,
|
||||
silence_ms=int(silence_ms),
|
||||
crossfade_duration=0.1,
|
||||
sample_rate=sample_rate,
|
||||
text_length=len(" ".join(text_chunks)),
|
||||
original_text=original_text,
|
||||
text_chunks=text_chunks,
|
||||
return_info=return_info,
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
# Compatibility with integration code that uses an all-caps acronym.
|
||||
AudioCPPProcessor = AudioCppProcessor
|
||||
@@ -0,0 +1,238 @@
|
||||
"""SRT timing orchestration for audio.cpp with a response-defined sample rate."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import os
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import torch
|
||||
|
||||
from utils.system.import_manager import import_manager
|
||||
from utils.timing.assembly import AudioAssemblyEngine
|
||||
from utils.timing.engine import TimingEngine
|
||||
from utils.timing.overlap_detection import SRTOverlapHandler
|
||||
from utils.timing.reporting import SRTReportGenerator
|
||||
|
||||
|
||||
def _processor_class():
|
||||
"""Load by path because this project also has a top-level ``nodes.py`` module."""
|
||||
path = os.path.join(os.path.dirname(__file__), "audio_cpp_processor.py")
|
||||
spec = importlib.util.spec_from_file_location("audio_cpp_processor_module", path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise ImportError(f"Cannot load audio.cpp processor from {path}")
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module.AudioCppProcessor
|
||||
|
||||
|
||||
def _adapter_class():
|
||||
"""Load directly so unrelated optional adapters are not imported eagerly."""
|
||||
path = os.path.abspath(
|
||||
os.path.join(os.path.dirname(__file__), "..", "..", "engines", "adapters", "audio_cpp_adapter.py")
|
||||
)
|
||||
spec = importlib.util.spec_from_file_location("audio_cpp_adapter_module", path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise ImportError(f"Cannot load audio.cpp adapter from {path}")
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module.AudioCppEngineAdapter
|
||||
|
||||
|
||||
class AudioCppSRTProcessor:
|
||||
"""Generate one subtitle cue at a time and assemble it on the SRT timeline."""
|
||||
|
||||
def __init__(self, node_instance: Any, config: Optional[Dict[str, Any]] = None):
|
||||
self.node_instance = node_instance
|
||||
self.config = dict(config or {})
|
||||
self.adapter = _adapter_class()(self.config)
|
||||
self._processor = _processor_class()(self.adapter, self.config)
|
||||
success, modules, message = import_manager.import_srt_modules()
|
||||
if not success or modules.get("SRTParser") is None:
|
||||
raise ImportError(f"audio.cpp SRT unavailable: {message}")
|
||||
self.SRTParser = modules["SRTParser"]
|
||||
|
||||
@property
|
||||
def processor(self) -> Any:
|
||||
return self._processor
|
||||
|
||||
@property
|
||||
def sample_rate(self) -> Optional[int]:
|
||||
return self.processor.sample_rate
|
||||
|
||||
def update_config(self, config: Optional[Dict[str, Any]]) -> None:
|
||||
self.config = dict(config or {})
|
||||
self.processor.update_config(self.config)
|
||||
|
||||
@staticmethod
|
||||
def _check_interrupt(index: Optional[int] = None, total: Optional[int] = None) -> None:
|
||||
try:
|
||||
import comfy.model_management as model_management
|
||||
|
||||
if getattr(model_management, "interrupt_processing", False) is True:
|
||||
location = f" at subtitle {index + 1}/{total}" if index is not None else ""
|
||||
raise InterruptedError(f"audio.cpp SRT generation interrupted{location}")
|
||||
except ImportError:
|
||||
return
|
||||
|
||||
@staticmethod
|
||||
def _adjustment(index: int, subtitle: Any, audio: torch.Tensor, sample_rate: int) -> Dict[str, Any]:
|
||||
natural = audio.shape[-1] / sample_rate
|
||||
target = float(subtitle.duration)
|
||||
ratio = target / natural if natural > 0 else 1.0
|
||||
return {
|
||||
"index": index,
|
||||
"segment_index": index,
|
||||
"sequence": subtitle.sequence,
|
||||
"natural_duration": natural,
|
||||
"target_start": subtitle.start_time,
|
||||
"target_end": subtitle.end_time,
|
||||
"target_duration": target,
|
||||
"start_time": subtitle.start_time,
|
||||
"end_time": subtitle.end_time,
|
||||
"stretch_factor": ratio,
|
||||
"needs_stretching": abs(ratio - 1.0) > 0.05,
|
||||
"stretch_type": "compress" if ratio < 1 else "expand" if ratio > 1 else "none",
|
||||
"adjustment": natural - target,
|
||||
"adjusted_start": subtitle.start_time,
|
||||
"adjusted_end": subtitle.end_time,
|
||||
"adjusted_duration": natural,
|
||||
}
|
||||
|
||||
def process_srt_content(
|
||||
self,
|
||||
srt_content: str,
|
||||
voice_mapping: Optional[Dict[str, Any]],
|
||||
seed: int,
|
||||
timing_mode: str,
|
||||
timing_params: Optional[Dict[str, Any]],
|
||||
enable_audio_cache: bool = True,
|
||||
) -> Tuple[Dict[str, Any], str, str, str]:
|
||||
self._check_interrupt()
|
||||
subtitles = self.SRTParser().parse_srt_content(srt_content, allow_overlaps=True)
|
||||
if not subtitles:
|
||||
raise ValueError("audio.cpp SRT input contains no subtitles")
|
||||
|
||||
has_overlaps = SRTOverlapHandler.detect_overlaps(subtitles)
|
||||
active_mode, switched = SRTOverlapHandler.handle_smart_natural_fallback(
|
||||
timing_mode, has_overlaps, "audio.cpp SRT"
|
||||
)
|
||||
self.processor.reset_sample_rate()
|
||||
audio_segments: List[Optional[torch.Tensor]] = []
|
||||
for index, subtitle in enumerate(subtitles):
|
||||
self._check_interrupt(index, len(subtitles))
|
||||
text = str(subtitle.text or "").strip()
|
||||
if not text:
|
||||
audio_segments.append(None)
|
||||
continue
|
||||
records = self.processor.process_text(
|
||||
text=text,
|
||||
voice_mapping=voice_mapping or {},
|
||||
seed=int(seed) + index,
|
||||
enable_chunking=False,
|
||||
enable_audio_cache=enable_audio_cache,
|
||||
apply_edit_postprocessing=True,
|
||||
show_text_logging=True,
|
||||
reset_sample_rate=False,
|
||||
)
|
||||
if not records:
|
||||
raise RuntimeError(f"audio.cpp produced no audio for subtitle {index + 1}")
|
||||
audio = self.processor.combine_audio_segments(
|
||||
records, method="auto", silence_ms=0, original_text=text
|
||||
)
|
||||
if audio.dim() == 1:
|
||||
audio = audio.unsqueeze(0)
|
||||
elif audio.dim() == 3 and audio.shape[0] == 1:
|
||||
audio = audio.squeeze(0)
|
||||
audio_segments.append(audio.detach().to(device="cpu", dtype=torch.float32))
|
||||
|
||||
sample_rate = self.processor.sample_rate
|
||||
if sample_rate is None:
|
||||
raise ValueError("audio.cpp could not determine a sample rate from the SRT content")
|
||||
completed_segments: List[torch.Tensor] = []
|
||||
for subtitle, audio in zip(subtitles, audio_segments):
|
||||
if audio is None:
|
||||
audio = torch.zeros(1, int(float(subtitle.duration) * sample_rate), dtype=torch.float32)
|
||||
completed_segments.append(audio)
|
||||
|
||||
adjustments = [
|
||||
self._adjustment(index, subtitle, completed_segments[index], sample_rate)
|
||||
for index, subtitle in enumerate(subtitles)
|
||||
]
|
||||
self._check_interrupt()
|
||||
final_audio, replacement, stretch_method = self._assemble(
|
||||
completed_segments, subtitles, active_mode, dict(timing_params or {}), sample_rate
|
||||
)
|
||||
if replacement is not None:
|
||||
adjustments = replacement
|
||||
|
||||
reporter = SRTReportGenerator()
|
||||
report = reporter.generate_timing_report(
|
||||
subtitles,
|
||||
adjustments,
|
||||
active_mode,
|
||||
has_overlaps,
|
||||
switched,
|
||||
timing_mode if switched else None,
|
||||
stretch_method,
|
||||
)
|
||||
adjusted_srt = reporter.generate_adjusted_srt_string(subtitles, adjustments, active_mode)
|
||||
if final_audio.dim() == 1:
|
||||
final_audio = final_audio.unsqueeze(0).unsqueeze(0)
|
||||
elif final_audio.dim() == 2:
|
||||
final_audio = final_audio.unsqueeze(0)
|
||||
duration = final_audio.shape[-1] / sample_rate
|
||||
mode_info = f"{active_mode} (switched from {timing_mode})" if switched else active_mode
|
||||
info = (
|
||||
f"Generated {duration:.1f}s audio.cpp SRT audio from {len(subtitles)} subtitles "
|
||||
f"using {mode_info} mode at {sample_rate} Hz"
|
||||
)
|
||||
return {"waveform": final_audio, "sample_rate": sample_rate}, info, report, adjusted_srt
|
||||
|
||||
@staticmethod
|
||||
def _assemble(
|
||||
audio_segments: List[torch.Tensor],
|
||||
subtitles: List[Any],
|
||||
mode: str,
|
||||
params: Dict[str, Any],
|
||||
sample_rate: int,
|
||||
):
|
||||
fade = params.get("fade_for_StretchToFit", 0.01)
|
||||
if mode == "stretch_to_fit":
|
||||
from engines.chatterbox.audio_timing import TimedAudioAssembler
|
||||
|
||||
assembler = TimedAudioAssembler(sample_rate)
|
||||
audio, method = assembler.assemble_timed_audio(
|
||||
audio_segments,
|
||||
[(item.start_time, item.end_time) for item in subtitles],
|
||||
fade_duration=fade,
|
||||
)
|
||||
return audio, None, method
|
||||
|
||||
assembler = AudioAssemblyEngine(sample_rate)
|
||||
if mode == "pad_with_silence":
|
||||
audio = assembler.assemble_with_overlaps(audio_segments, subtitles, torch.device("cpu"))
|
||||
return audio, None, None
|
||||
|
||||
timing = TimingEngine(sample_rate)
|
||||
if mode == "concatenate":
|
||||
replacements = timing.calculate_concatenation_adjustments(audio_segments, subtitles)
|
||||
audio = assembler.assemble_concatenation(audio_segments, fade)
|
||||
return audio, replacements, None
|
||||
|
||||
replacements, processed = timing.calculate_smart_timing_adjustments(
|
||||
audio_segments,
|
||||
subtitles,
|
||||
params.get("timing_tolerance", 2.0),
|
||||
params.get("max_stretch_ratio", 1.0),
|
||||
params.get("min_stretch_ratio", 0.5),
|
||||
torch.device("cpu"),
|
||||
)
|
||||
audio = assembler.assemble_smart_natural(
|
||||
audio_segments, processed, replacements, subtitles, torch.device("cpu")
|
||||
)
|
||||
return audio, replacements, None
|
||||
|
||||
|
||||
AudioCppSubtitleProcessor = AudioCppSRTProcessor
|
||||
AudioCPPSRTProcessor = AudioCppSRTProcessor
|
||||
@@ -0,0 +1,444 @@
|
||||
"""ComfyUI configuration node for the generic audio.cpp backend."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict, List, Mapping, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
|
||||
class AnyType(str):
|
||||
def __ne__(self, __value: object) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
any_type = AnyType("*")
|
||||
|
||||
|
||||
def _catalog_module():
|
||||
try:
|
||||
from utils.audio_cpp import catalog
|
||||
|
||||
return catalog
|
||||
except ImportError:
|
||||
return None
|
||||
|
||||
|
||||
def _fallback_specs() -> List[Dict[str, Any]]:
|
||||
root = os.path.abspath(
|
||||
os.path.join(os.path.dirname(__file__), "..", "..", "utils", "audio_cpp", "model_specs")
|
||||
)
|
||||
specs = []
|
||||
for path in glob.glob(os.path.join(root, "*.json")):
|
||||
try:
|
||||
with open(path, "r", encoding="utf-8") as handle:
|
||||
value = json.load(handle)
|
||||
if isinstance(value, dict) and value.get("family"):
|
||||
specs.append(value)
|
||||
except (OSError, json.JSONDecodeError):
|
||||
continue
|
||||
return specs
|
||||
|
||||
|
||||
def _family_choices() -> List[str]:
|
||||
catalog = _catalog_module()
|
||||
if catalog is not None and callable(getattr(catalog, "family_choices", None)):
|
||||
choices = list(catalog.family_choices())
|
||||
else:
|
||||
choices = [spec["family"] for spec in _fallback_specs()]
|
||||
choices = sorted({str(choice) for choice in choices if str(choice).strip()})
|
||||
return choices or ["qwen3_tts"]
|
||||
|
||||
|
||||
def _package_choices() -> List[str]:
|
||||
catalog = _catalog_module()
|
||||
if catalog is not None and callable(getattr(catalog, "package_choices", None)):
|
||||
choices = list(catalog.package_choices())
|
||||
else:
|
||||
choices = [
|
||||
package.get("id")
|
||||
for spec in _fallback_specs()
|
||||
for package in spec.get("packages", [])
|
||||
if isinstance(package, dict)
|
||||
]
|
||||
return ["auto"] + sorted({str(choice) for choice in choices if choice})
|
||||
|
||||
|
||||
def _recommended_package(family: str) -> str:
|
||||
catalog = _catalog_module()
|
||||
if catalog is not None and callable(getattr(catalog, "recommended_package", None)):
|
||||
value = catalog.recommended_package(family)
|
||||
if value:
|
||||
return str(value)
|
||||
for spec in _fallback_specs():
|
||||
if spec.get("family") != family:
|
||||
continue
|
||||
recommended = (spec.get("ui") or {}).get("recommended_package")
|
||||
if recommended:
|
||||
return str(recommended)
|
||||
for package in spec.get("packages", []):
|
||||
if package.get("default"):
|
||||
return str(package["id"])
|
||||
return "auto"
|
||||
|
||||
|
||||
def _resolve_task(family: str, package_id: str, requested: str) -> str:
|
||||
requested = str(requested or "auto").lower()
|
||||
if requested in {"tts", "clon", "vdes", "vc", "s2s", "svc", "asr", "diar"}:
|
||||
return requested
|
||||
catalog = _catalog_module()
|
||||
if catalog is not None and callable(getattr(catalog, "resolve_task", None)):
|
||||
return str(catalog.resolve_task(family, package_id, requested="auto")).lower()
|
||||
package_lower = package_id.lower()
|
||||
if "voicedesign" in package_lower or "voice_design" in package_lower:
|
||||
return "vdes"
|
||||
if family in {"chatterbox", "confucius4_tts"}:
|
||||
return "clon"
|
||||
return "tts"
|
||||
|
||||
|
||||
def _validate_package(family: str, package_id: str) -> None:
|
||||
catalog = _catalog_module()
|
||||
getter = getattr(catalog, "get_package", None) if catalog is not None else None
|
||||
if not callable(getter) or package_id == "auto":
|
||||
return
|
||||
value = getter(package_id)
|
||||
if value is None:
|
||||
raise ValueError(f"Unknown audio.cpp package: {package_id}")
|
||||
package_family = value.get("family") if isinstance(value, Mapping) else getattr(value, "family", None)
|
||||
if package_family and str(package_family) != family:
|
||||
raise ValueError(f"audio.cpp package '{package_id}' does not belong to family '{family}'")
|
||||
|
||||
|
||||
class AudioCppEngineNode:
|
||||
"""Describe either a managed audio.cpp runtime or an existing installation."""
|
||||
|
||||
@classmethod
|
||||
def NAME(cls):
|
||||
return "⚙️ audio.cpp Multi-TTS Engine"
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls):
|
||||
families = _family_choices()
|
||||
default_family = "qwen3_tts" if "qwen3_tts" in families else families[0]
|
||||
packages = _package_choices()
|
||||
return {
|
||||
"required": {
|
||||
"connection_mode": (
|
||||
["auto", "external_server", "existing_binary", "managed"],
|
||||
{
|
||||
"default": "auto",
|
||||
"tooltip": "Auto prefers a supplied server or binary, then the suite-managed runtime.",
|
||||
},
|
||||
),
|
||||
"family": (
|
||||
families,
|
||||
{
|
||||
"default": default_family,
|
||||
"tooltip": "audio.cpp model family. The package list and capability panel update to match this selection.",
|
||||
},
|
||||
),
|
||||
"package_id": (
|
||||
packages,
|
||||
{
|
||||
"default": "auto",
|
||||
"tooltip": "Auto selects the pinned recommended package for the chosen family.",
|
||||
},
|
||||
),
|
||||
"task": (
|
||||
["auto", "tts", "clon", "vdes", "vc", "s2s", "svc", "asr", "diar"],
|
||||
{
|
||||
"default": "auto",
|
||||
"tooltip": "Runtime task. Auto lets the connected unified node use the family's normal task; choose an explicit task only for advanced routing or external-server matching.",
|
||||
},
|
||||
),
|
||||
"backend": (
|
||||
["auto", "cuda", "cpu", "vulkan", "metal", "hip"],
|
||||
{
|
||||
"default": "auto",
|
||||
"tooltip": "Native audio.cpp compute backend. Auto selects an installed CUDA runtime when available, otherwise CPU.",
|
||||
},
|
||||
),
|
||||
"device": (
|
||||
"INT",
|
||||
{
|
||||
"default": 0,
|
||||
"min": 0,
|
||||
"max": 31,
|
||||
"tooltip": "Zero-based native device index. Keep 0 unless using another GPU/device.",
|
||||
},
|
||||
),
|
||||
"threads": (
|
||||
"INT",
|
||||
{
|
||||
"default": 4,
|
||||
"min": 1,
|
||||
"max": 128,
|
||||
"tooltip": "Native backend/OpenMP workers. Four matches the audio.cpp CLI default; tune for your CPU.",
|
||||
},
|
||||
),
|
||||
"language": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "auto",
|
||||
"tooltip": "Language code passed to audio.cpp. Auto lets the selected model infer or use its default language.",
|
||||
},
|
||||
),
|
||||
},
|
||||
"optional": {
|
||||
"server_url": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "",
|
||||
"tooltip": "Required only for external_server mode, for example http://127.0.0.1:8080.",
|
||||
},
|
||||
),
|
||||
"binary_path": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "",
|
||||
"tooltip": "Optional path to an existing audiocpp_server executable. Leave blank to use the Suite-managed runtime.",
|
||||
},
|
||||
),
|
||||
"model_path": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "",
|
||||
"tooltip": "Optional existing audio.cpp model/package directory. Leave blank for discovery or managed download.",
|
||||
},
|
||||
),
|
||||
"model_id": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "",
|
||||
"tooltip": "Server model identifier. Usually leave blank; required when an external server exposes multiple models.",
|
||||
},
|
||||
),
|
||||
"voice_id": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "",
|
||||
"tooltip": "Optional built-in voice/preset ID for families such as Supertonic. Reference audio takes precedence when supported.",
|
||||
},
|
||||
),
|
||||
"instruct": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "",
|
||||
"multiline": True,
|
||||
"tooltip": "Optional natural-language voice design or style instruction. Used only by families/tasks that support instructions.",
|
||||
},
|
||||
),
|
||||
"speaker2": (any_type, {"tooltip": "Optional ordered character/Speaker 2 reference."}),
|
||||
"temperature": ("FLOAT", {"default": -1.0, "min": -1.0, "max": 5.0, "step": 0.05, "tooltip": "Sampling temperature. -1 uses the selected model/package default."}),
|
||||
"top_p": ("FLOAT", {"default": -1.0, "min": -1.0, "max": 1.0, "step": 0.01, "tooltip": "Nucleus sampling threshold. -1 uses the model default."}),
|
||||
"top_k": ("INT", {"default": -1, "min": -1, "max": 1000, "tooltip": "Top-k sampling limit. -1 uses the model default."}),
|
||||
"repetition_penalty": (
|
||||
"FLOAT",
|
||||
{"default": -1.0, "min": -1.0, "max": 5.0, "step": 0.05, "tooltip": "Token repetition penalty. -1 uses the model default."},
|
||||
),
|
||||
"max_tokens": ("INT", {"default": 0, "min": 0, "max": 131072, "tooltip": "Maximum generated tokens. 0 lets the model choose its normal limit."}),
|
||||
"max_steps": ("INT", {"default": 0, "min": 0, "max": 4096, "tooltip": "Maximum generation/decoder steps where supported. 0 uses the model default."}),
|
||||
"num_inference_steps": ("INT", {"default": 0, "min": 0, "max": 1000, "tooltip": "Flow/diffusion inference steps where supported. 0 uses the model default."}),
|
||||
"guidance_scale": (
|
||||
"FLOAT",
|
||||
{"default": -1.0, "min": -1.0, "max": 100.0, "step": 0.05, "tooltip": "Classifier-free guidance scale where supported. -1 uses the model default."},
|
||||
),
|
||||
"advanced_json": (
|
||||
"STRING",
|
||||
{
|
||||
"default": "{}",
|
||||
"multiline": True,
|
||||
"tooltip": "Model-specific audio.cpp request options as a JSON object.",
|
||||
},
|
||||
),
|
||||
"auto_download_runtime": (
|
||||
"BOOLEAN",
|
||||
{
|
||||
"default": True,
|
||||
"tooltip": "Automatically install the pinned audio.cpp runtime into Suite-managed storage when no usable runtime is found. Existing external binaries are never copied.",
|
||||
},
|
||||
),
|
||||
"auto_download_model": (
|
||||
"BOOLEAN",
|
||||
{
|
||||
"default": True,
|
||||
"tooltip": "Automatically download the selected audio.cpp package into models/TTS/audio.cpp/models when it is not already available. Downloads use direct files, not the Hugging Face cache.",
|
||||
},
|
||||
),
|
||||
"show_server_console": (
|
||||
"BOOLEAN",
|
||||
{
|
||||
"default": False,
|
||||
"tooltip": "Debug only: launch a visible console for a Suite-owned audio.cpp server.",
|
||||
},
|
||||
),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TTS_ENGINE",)
|
||||
RETURN_NAMES = ("TTS_engine",)
|
||||
FUNCTION = "create_engine_config"
|
||||
CATEGORY = "TTS Audio Suite/⚙️ Engines"
|
||||
|
||||
def create_engine_config(
|
||||
self,
|
||||
connection_mode: str,
|
||||
family: str,
|
||||
package_id: str,
|
||||
task: str,
|
||||
backend: str,
|
||||
device: int,
|
||||
threads: int,
|
||||
language: str,
|
||||
server_url: str = "",
|
||||
binary_path: str = "",
|
||||
model_path: str = "",
|
||||
model_id: str = "",
|
||||
voice_id: str = "",
|
||||
instruct: str = "",
|
||||
temperature: float = -1.0,
|
||||
top_p: float = -1.0,
|
||||
top_k: int = -1,
|
||||
repetition_penalty: float = -1.0,
|
||||
max_tokens: int = 0,
|
||||
max_steps: int = 0,
|
||||
num_inference_steps: int = 0,
|
||||
guidance_scale: float = -1.0,
|
||||
advanced_json: str = "{}",
|
||||
auto_download_runtime: bool = True,
|
||||
auto_download_model: bool = True,
|
||||
show_server_console: bool = False,
|
||||
speaker_mode: str = "Custom Character Switching",
|
||||
speaker2: Any = None,
|
||||
**kwargs: Any,
|
||||
) -> tuple:
|
||||
mode = str(connection_mode).strip().lower()
|
||||
if mode not in {"auto", "external_server", "existing_binary", "managed"}:
|
||||
raise ValueError(f"Unsupported audio.cpp connection mode: {connection_mode}")
|
||||
family = str(family).strip()
|
||||
package_id = str(package_id or "auto").strip()
|
||||
if not family:
|
||||
raise ValueError("audio.cpp family is required")
|
||||
|
||||
url = str(server_url or "").strip().rstrip("/")
|
||||
binary = os.path.abspath(os.path.expanduser(binary_path)) if binary_path.strip() else ""
|
||||
model = os.path.abspath(os.path.expanduser(model_path)) if model_path.strip() else ""
|
||||
if mode == "external_server":
|
||||
parsed = urlparse(url)
|
||||
if parsed.scheme not in {"http", "https"} or not parsed.netloc:
|
||||
raise ValueError("audio.cpp external_server mode requires a valid HTTP(S) server_url")
|
||||
if mode == "existing_binary":
|
||||
if not binary:
|
||||
raise ValueError("audio.cpp existing_binary mode requires binary_path")
|
||||
if not os.path.isfile(binary):
|
||||
raise FileNotFoundError(f"audio.cpp binary not found: {binary}")
|
||||
|
||||
try:
|
||||
advanced = json.loads(advanced_json or "{}")
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"Invalid audio.cpp advanced JSON: {exc.msg}") from exc
|
||||
if not isinstance(advanced, Mapping):
|
||||
raise ValueError("audio.cpp advanced JSON must contain an object")
|
||||
|
||||
uses_existing_server = mode == "external_server"
|
||||
if package_id == "auto" and not uses_existing_server:
|
||||
package_id = _recommended_package(family)
|
||||
if not uses_existing_server:
|
||||
_validate_package(family, package_id)
|
||||
resolved_task = _resolve_task(family, package_id, task)
|
||||
else:
|
||||
# The loaded model reported by /v1/models owns this decision.
|
||||
resolved_task = str(task or "auto").lower()
|
||||
|
||||
config: Dict[str, Any] = {
|
||||
"engine_type": "audio_cpp",
|
||||
"connection_mode": mode,
|
||||
"family": family,
|
||||
"package_id": package_id,
|
||||
"requested_task": str(task or "auto").lower(),
|
||||
"task": resolved_task,
|
||||
"backend": str(backend).lower(),
|
||||
"device": int(device),
|
||||
"threads": int(threads),
|
||||
"language": str(language or "auto"),
|
||||
"server_url": url,
|
||||
"external_server_url": url,
|
||||
"binary_path": binary,
|
||||
"model_path": model,
|
||||
"model_id": str(model_id or "").strip(),
|
||||
"voice_id": str(voice_id or "").strip(),
|
||||
"instruct": str(instruct or "").strip(),
|
||||
"advanced_options": dict(advanced),
|
||||
"auto_download_runtime": bool(auto_download_runtime),
|
||||
"auto_download_model": bool(auto_download_model),
|
||||
"show_server_console": bool(show_server_console),
|
||||
"multi_speaker_mode": str(speaker_mode),
|
||||
}
|
||||
speakers = [speaker2] if speaker2 is not None else []
|
||||
dynamic_speakers = []
|
||||
for key, value in kwargs.items():
|
||||
if key.startswith("speaker") and key[7:].isdigit() and value is not None:
|
||||
dynamic_speakers.append((int(key[7:]), value))
|
||||
speakers.extend(value for _, value in sorted(dynamic_speakers))
|
||||
config["speaker_references"] = speakers
|
||||
|
||||
try:
|
||||
from utils.audio_cpp.capabilities import get_capability
|
||||
|
||||
capability = get_capability(family)
|
||||
maximum = int(capability["native_multi_speaker"]["max_speakers"])
|
||||
if len(speakers) > max(0, maximum - 1):
|
||||
raise ValueError(f"audio.cpp {family} supports at most {maximum} speakers")
|
||||
if speaker_mode == "Native Multi-Speaker" and capability["native_multi_speaker"]["suite_status"] != "supported":
|
||||
raise ValueError(
|
||||
f"audio.cpp {family} native multi-speaker mode is not integrated; "
|
||||
"use Custom Character Switching"
|
||||
)
|
||||
except ImportError:
|
||||
pass
|
||||
optional_values = {
|
||||
"temperature": float(temperature),
|
||||
"top_p": float(top_p),
|
||||
"top_k": int(top_k),
|
||||
"repetition_penalty": float(repetition_penalty),
|
||||
"guidance_scale": float(guidance_scale),
|
||||
}
|
||||
for key, value in optional_values.items():
|
||||
if value >= 0:
|
||||
config[key] = value
|
||||
for key, value in {
|
||||
"max_tokens": int(max_tokens),
|
||||
"max_steps": int(max_steps),
|
||||
"num_inference_steps": int(num_inference_steps),
|
||||
}.items():
|
||||
if value > 0:
|
||||
config[key] = value
|
||||
|
||||
try:
|
||||
from utils.audio_cpp.capabilities import get_capability as load_capability
|
||||
|
||||
family_capability = load_capability(family)
|
||||
suite_tasks = set(family_capability.get("suite_tasks", []))
|
||||
except (ImportError, KeyError, ValueError):
|
||||
suite_tasks = {"tts"}
|
||||
capabilities = []
|
||||
if "tts" in suite_tasks:
|
||||
capabilities.append("tts")
|
||||
if "asr" in suite_tasks:
|
||||
capabilities.append("asr")
|
||||
if "voice_conversion" in suite_tasks:
|
||||
capabilities.append("voice_conversion")
|
||||
if "diarization" in suite_tasks:
|
||||
capabilities.append("diarization")
|
||||
catalog_module = _catalog_module()
|
||||
family_record = catalog_module.get_family(family) if catalog_module is not None else None
|
||||
if resolved_task == "vdes" or "vdes" in getattr(family_record, "runtime_tasks", ()):
|
||||
capabilities.append("voice_design")
|
||||
return ({"engine_type": "audio_cpp", "config": config, "capabilities": capabilities},)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {"AudioCppEngineNode": AudioCppEngineNode}
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {"AudioCppEngineNode": "⚙️ audio.cpp Multi-TTS Engine"}
|
||||
@@ -48,7 +48,7 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
|
||||
return {
|
||||
"required": {
|
||||
"engine": ("TTS_ENGINE", {
|
||||
"tooltip": "ASR-capable engine configuration (for example Qwen3-TTS Engine or Granite ASR Engine). This node auto-routes to the correct ASR adapter based on the engine type."
|
||||
"tooltip": "ASR-capable engine configuration. Supports Qwen3-TTS ASR, Granite ASR, and audio.cpp families whose capability panel shows ASR. The unified node routes to the correct adapter and preserves available timing/speaker data."
|
||||
}),
|
||||
"audio": (any_typ, {
|
||||
"tooltip": "Audio to transcribe. Accepts AUDIO, Character Voices output, or VideoHelper audio."
|
||||
@@ -96,7 +96,7 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
|
||||
}),
|
||||
"timestamps": (["none", "word"], {
|
||||
"default": "none",
|
||||
"tooltip": "Timing detail for the ASR timing output:\n• none: Text only, no reusable timed words/segments\n• word: Word-level timings for timestamp-capable ASR paths\n\nUse word timings if you plan to feed this into the Text to SRT Builder.\n\nGranite note: word timestamps are native on the plus model variant when diarization is off. Other Granite timestamp paths use the separate Qwen forced aligner."
|
||||
"tooltip": "Timing detail for the ASR timing output:\n• none: Text only, except native speaker turns may still carry segment timing\n• word: Request or preserve word timings when the selected ASR family supports them\n\nUse word timings for Text to SRT Builder.\n\nGranite: the plus model has native timestamps; other variants use the Qwen forced aligner.\naudio.cpp: native words/segments are preserved. Qwen3-ASR specifically needs its optional forced-aligner model for requested word timings and will otherwise continue with text only."
|
||||
}),
|
||||
"chunk_size": ("INT", {
|
||||
"default": 30, "min": 0, "max": 600, "step": 1,
|
||||
@@ -112,7 +112,7 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
|
||||
}),
|
||||
"diarization": ("BOOLEAN", {
|
||||
"default": False,
|
||||
"tooltip": "Speaker Diarization (Speaker Attribution):\n• True: Attribute speech to speakers if supported (for example [Speaker 1] hello)\n• False: Plain transcription without speaker turns\n\nGranite note: Native speaker attribution is supported on the 'plus' model variant. If combined with word-level timestamps, the system automatically uses the Qwen forced aligner to time-align the speakers' words."
|
||||
"tooltip": "Speaker attribution:\n• True: Preserve speaker turns when the selected ASR engine returns them\n• False: Return plain transcription/timing\n\nGranite 4.1 plus and audio.cpp VibeVoice-ASR provide native speaker attribution. Other audio.cpp ASR families return a warning instead of inventing speaker labels."
|
||||
}),
|
||||
}
|
||||
}
|
||||
@@ -171,6 +171,13 @@ class UnifiedASRTranscribeNode(BaseChatterBoxNode):
|
||||
engine_cfg = engine.get("config", engine)
|
||||
cache_data = {
|
||||
"engine_type": engine.get("engine_type"),
|
||||
"family": engine_cfg.get("family"),
|
||||
"package_id": engine_cfg.get("package_id"),
|
||||
"model_id": engine_cfg.get("model_id"),
|
||||
"model_path": engine_cfg.get("model_path"),
|
||||
"connection_mode": engine_cfg.get("connection_mode"),
|
||||
"server_url": engine_cfg.get("server_url"),
|
||||
"advanced_options": str(engine_cfg.get("advanced_options", {})),
|
||||
"model_name": engine_cfg.get("model_name"),
|
||||
"model_size": engine_cfg.get("model_size"),
|
||||
"device": engine_cfg.get("device"),
|
||||
|
||||
@@ -254,6 +254,16 @@ Hello! This is unified SRT TTS with character switching.
|
||||
stable_params['dtype'] = config.get('dtype', 'auto')
|
||||
stable_params['attention'] = config.get('attention', 'auto')
|
||||
|
||||
if engine_type == "audio_cpp":
|
||||
for key in (
|
||||
'connection_mode', 'server_url', 'server_model_id', 'model_id',
|
||||
'binary_path', 'model_path', 'model_roots', 'family',
|
||||
'package_id', 'task', 'backend', 'device', 'device_index',
|
||||
'threads', 'model_spec_override', 'load_options',
|
||||
'session_options', 'show_server_console',
|
||||
):
|
||||
stable_params[key] = config.get(key)
|
||||
|
||||
# IndexTTS 2.0 and 2.5 are distinct checkpoints/backends. Every
|
||||
# load-time option must participate in the processor cache key or
|
||||
# changing the engine node can silently keep the old adapter alive.
|
||||
@@ -995,6 +1005,38 @@ Hello! This is unified SRT TTS with character switching.
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
processor_path = os.path.join(nodes_dir, "audio_cpp", "audio_cpp_srt_processor.py")
|
||||
processor_spec = importlib.util.spec_from_file_location(
|
||||
"audio_cpp_srt_processor_module", processor_path
|
||||
)
|
||||
if processor_spec is None or processor_spec.loader is None:
|
||||
raise ImportError(f"Cannot load audio.cpp SRT processor from {processor_path}")
|
||||
processor_module = importlib.util.module_from_spec(processor_spec)
|
||||
processor_spec.loader.exec_module(processor_module)
|
||||
AudioCppSRTProcessor = processor_module.AudioCppSRTProcessor
|
||||
|
||||
class AudioCppSRTWrapper:
|
||||
def __init__(self, cfg):
|
||||
self.config = cfg.copy()
|
||||
self.processor = AudioCppSRTProcessor(self, self.config)
|
||||
|
||||
def update_config(self, new_config):
|
||||
self.config = new_config.copy()
|
||||
self.processor.update_config(self.config)
|
||||
|
||||
def check_interrupt(self):
|
||||
if model_management.interrupt_processing:
|
||||
raise InterruptedError("audio.cpp SRT processing interrupted by user")
|
||||
|
||||
engine_instance = AudioCppSRTWrapper(config)
|
||||
import time
|
||||
self._cached_engine_instances[cache_key] = {
|
||||
'instance': engine_instance,
|
||||
'timestamp': time.time(),
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
else:
|
||||
raise ValueError(f"Unknown engine type: {engine_type}")
|
||||
|
||||
@@ -1146,6 +1188,11 @@ Hello! This is unified SRT TTS with character switching.
|
||||
|
||||
if not engine_type:
|
||||
raise ValueError("TTS engine missing engine_type")
|
||||
capabilities = TTS_engine.get("capabilities", [])
|
||||
if capabilities and "tts" not in capabilities:
|
||||
raise ValueError(
|
||||
f"Engine '{engine_type}' does not support TTS/SRT. Connect it to its compatible unified node."
|
||||
)
|
||||
|
||||
if config.get("model_role") == "voice_design":
|
||||
selected_model = config.get("model_variant") or config.get("model_name") or "selected model"
|
||||
@@ -1660,6 +1707,29 @@ Hello! This is unified SRT TTS with character switching.
|
||||
timing_params=timing_params
|
||||
)
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
voice_mapping = {}
|
||||
if audio_tensor is not None or audio_path:
|
||||
voice_mapping['narrator'] = {
|
||||
'audio': audio_tensor,
|
||||
'audio_path': audio_path,
|
||||
'reference_text': reference_text or '',
|
||||
}
|
||||
timing_params = {
|
||||
'fade_for_StretchToFit': fade_for_StretchToFit,
|
||||
'max_stretch_ratio': max_stretch_ratio,
|
||||
'min_stretch_ratio': min_stretch_ratio,
|
||||
'timing_tolerance': timing_tolerance,
|
||||
}
|
||||
result = engine_instance.processor.process_srt_content(
|
||||
srt_content=srt_content,
|
||||
voice_mapping=voice_mapping,
|
||||
seed=seed,
|
||||
timing_mode=timing_mode,
|
||||
timing_params=timing_params,
|
||||
enable_audio_cache=enable_audio_cache,
|
||||
)
|
||||
|
||||
else:
|
||||
raise ValueError(f"Unknown engine type: {engine_type}")
|
||||
|
||||
@@ -1702,6 +1772,8 @@ Hello! This is unified SRT TTS with character switching.
|
||||
or "MOSS-TTSD Native Multi-Speaker Dialogue does not support this SRT input" in msg
|
||||
):
|
||||
raise
|
||||
if engine_type == "audio_cpp":
|
||||
raise
|
||||
if isinstance(e, InterruptedError):
|
||||
raise
|
||||
error_msg = f"❌ TTS SRT generation failed: {e}"
|
||||
|
||||
@@ -250,6 +250,19 @@ Back to the main narrator voice for the conclusion.""",
|
||||
stable_params['dtype'] = config.get('dtype', 'auto')
|
||||
stable_params['attention'] = config.get('attention', 'auto')
|
||||
|
||||
if engine_type == "audio_cpp":
|
||||
# audio.cpp owns a persistent native server. Everything that changes
|
||||
# that server/model session belongs in the instance cache identity;
|
||||
# request-time sampling controls deliberately do not.
|
||||
for key in (
|
||||
'connection_mode', 'server_url', 'server_model_id', 'model_id',
|
||||
'binary_path', 'model_path', 'model_roots', 'family',
|
||||
'package_id', 'task', 'backend', 'device', 'device_index',
|
||||
'threads', 'model_spec_override', 'load_options',
|
||||
'session_options', 'show_server_console',
|
||||
):
|
||||
stable_params[key] = config.get(key)
|
||||
|
||||
# IndexTTS 2.0 and 2.5 are distinct checkpoints/backends. Every
|
||||
# load-time option must participate in the processor cache key or
|
||||
# changing the engine node can silently keep the old adapter alive.
|
||||
@@ -746,6 +759,48 @@ Back to the main narrator voice for the conclusion.""",
|
||||
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
adapter_path = os.path.join(project_root, "engines", "adapters", "audio_cpp_adapter.py")
|
||||
adapter_spec = importlib.util.spec_from_file_location(
|
||||
"audio_cpp_adapter_module", adapter_path
|
||||
)
|
||||
if adapter_spec is None or adapter_spec.loader is None:
|
||||
raise ImportError(f"Cannot load audio.cpp adapter from {adapter_path}")
|
||||
adapter_module = importlib.util.module_from_spec(adapter_spec)
|
||||
adapter_spec.loader.exec_module(adapter_module)
|
||||
AudioCppEngineAdapter = adapter_module.AudioCppEngineAdapter
|
||||
processor_path = os.path.join(nodes_dir, "audio_cpp", "audio_cpp_processor.py")
|
||||
processor_spec = importlib.util.spec_from_file_location(
|
||||
"audio_cpp_processor_module", processor_path
|
||||
)
|
||||
if processor_spec is None or processor_spec.loader is None:
|
||||
raise ImportError(f"Cannot load audio.cpp processor from {processor_path}")
|
||||
processor_module = importlib.util.module_from_spec(processor_spec)
|
||||
processor_spec.loader.exec_module(processor_module)
|
||||
AudioCppProcessor = processor_module.AudioCppProcessor
|
||||
|
||||
class AudioCppWrapper:
|
||||
def __init__(self, cfg):
|
||||
self.config = cfg.copy()
|
||||
self.adapter = AudioCppEngineAdapter(self.config)
|
||||
self.processor = AudioCppProcessor(self.adapter, self.config)
|
||||
|
||||
def update_config(self, new_config):
|
||||
self.config = new_config.copy()
|
||||
self.processor.update_config(self.config)
|
||||
|
||||
def check_interrupt(self):
|
||||
if model_management.interrupt_processing:
|
||||
raise InterruptedError("audio.cpp processing interrupted by user")
|
||||
|
||||
engine_instance = AudioCppWrapper(config)
|
||||
import time
|
||||
self._cached_engine_instances[cache_key] = {
|
||||
'instance': engine_instance,
|
||||
'timestamp': time.time(),
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "step_audio_editx":
|
||||
# Create Step Audio EditX wrapper instance
|
||||
class StepAudioEditXWrapper:
|
||||
@@ -1010,6 +1065,11 @@ Back to the main narrator voice for the conclusion.""",
|
||||
|
||||
if not engine_type:
|
||||
raise ValueError("TTS engine missing engine_type")
|
||||
capabilities = TTS_engine.get("capabilities", [])
|
||||
if capabilities and "tts" not in capabilities:
|
||||
raise ValueError(
|
||||
f"Engine '{engine_type}' does not support TTS. Connect it to its compatible unified node."
|
||||
)
|
||||
|
||||
if config.get("model_role") == "voice_design":
|
||||
selected_model = config.get("model_variant") or config.get("model_name") or "selected model"
|
||||
@@ -2055,6 +2115,52 @@ Back to the main narrator voice for the conclusion.""",
|
||||
seed=seed
|
||||
)
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
import re
|
||||
|
||||
voice_mapping = {}
|
||||
if audio_tensor is not None or audio_path:
|
||||
voice_mapping['narrator'] = {
|
||||
'audio': audio_tensor,
|
||||
'audio_path': audio_path,
|
||||
'reference_text': reference_text or '',
|
||||
}
|
||||
|
||||
audio_segments = engine_instance.processor.process_text(
|
||||
text=text,
|
||||
voice_mapping=voice_mapping,
|
||||
seed=seed,
|
||||
enable_chunking=enable_chunking,
|
||||
max_chars_per_chunk=max_chars_per_chunk,
|
||||
enable_audio_cache=enable_audio_cache,
|
||||
)
|
||||
audio_result, chunk_info = engine_instance.processor.combine_audio_segments(
|
||||
segments=audio_segments,
|
||||
method=chunk_combination_method,
|
||||
silence_ms=silence_between_chunks_ms,
|
||||
original_text=text,
|
||||
return_info=True,
|
||||
)
|
||||
sample_rate = engine_instance.processor.sample_rate
|
||||
if not sample_rate:
|
||||
raise RuntimeError("audio.cpp returned no sample rate")
|
||||
clean_text = re.sub(r'\[.*?\]', '', text)
|
||||
duration = audio_result.shape[-1] / sample_rate if audio_result.numel() else 0.0
|
||||
family = config.get('family') or config.get('server_model_id') or 'external model'
|
||||
base_info = (
|
||||
f"Generated {duration:.1f}s audio from {len(clean_text)} characters "
|
||||
f"(audio.cpp {family}, {sample_rate} Hz, narrator: {char_display})"
|
||||
)
|
||||
base_info += "\n🎭 Character switching, pause tags, and per-segment parameters supported"
|
||||
from utils.audio.chunk_timing import ChunkTimingHelper
|
||||
generation_info = ChunkTimingHelper.enhance_generation_info(
|
||||
f"✅ {base_info}", chunk_info
|
||||
)
|
||||
result = (
|
||||
AudioProcessingUtils.format_for_comfyui(audio_result, sample_rate),
|
||||
generation_info,
|
||||
)
|
||||
|
||||
else:
|
||||
raise ValueError(f"Unknown engine type: {engine_type}")
|
||||
|
||||
@@ -2090,7 +2196,7 @@ Back to the main narrator voice for the conclusion.""",
|
||||
raise
|
||||
if "MOSS LoRA/base model mismatch" in str(e):
|
||||
raise
|
||||
if engine_type == "index_tts":
|
||||
if engine_type in {"index_tts", "audio_cpp"}:
|
||||
raise
|
||||
if isinstance(e, InterruptedError):
|
||||
raise
|
||||
|
||||
@@ -51,7 +51,7 @@ GLOBAL_RVC_ITERATION_CACHE = {}
|
||||
class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
"""
|
||||
Unified Voice Changer Node - Engine-agnostic voice conversion.
|
||||
Currently supports ChatterBox, prepared for future RVC and other voice conversion engines.
|
||||
Routes ChatterBox, CosyVoice, RVC, and compatible audio.cpp families.
|
||||
Replaces ChatterBox VC node with engine-agnostic architecture.
|
||||
"""
|
||||
|
||||
@@ -64,7 +64,7 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
return {
|
||||
"required": {
|
||||
"TTS_engine": ("TTS_ENGINE", {
|
||||
"tooltip": "TTS/VC engine configuration. Supports ChatterBox TTS Engine, CosyVoice Engine, and RVC Engine for voice conversion."
|
||||
"tooltip": "Engine configuration for source-to-target voice conversion. Supports ChatterBox, CosyVoice, RVC, and audio.cpp families whose panel shows Voice conversion (Chatterbox, VeVo2, or Seed-VC)."
|
||||
}),
|
||||
"source_audio": (any_typ, {
|
||||
"tooltip": "The original voice audio you want to convert to sound like the target voice. Accepts AUDIO input or Character Voices node output."
|
||||
@@ -574,7 +574,7 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "cosyvoice":
|
||||
elif engine_type == "cosyvoice":
|
||||
# Import and create the CosyVoice VC processor
|
||||
cosyvoice_vc_path = os.path.join(nodes_dir, "cosyvoice", "cosyvoice_vc_processor.py")
|
||||
cosyvoice_vc_spec = importlib.util.spec_from_file_location("cosyvoice_vc_module", cosyvoice_vc_path)
|
||||
@@ -589,10 +589,21 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
self._cached_engine_instances[cache_key] = {
|
||||
'instance': engine_instance,
|
||||
'timestamp': time.time()
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "f5tts":
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
from engines.adapters.audio_cpp_vc_adapter import AudioCppVoiceConversionAdapter
|
||||
|
||||
engine_instance = AudioCppVoiceConversionAdapter(config)
|
||||
import time
|
||||
self._cached_engine_instances[cache_key] = {
|
||||
'instance': engine_instance,
|
||||
'timestamp': time.time()
|
||||
}
|
||||
return engine_instance
|
||||
|
||||
elif engine_type == "f5tts":
|
||||
# F5-TTS doesn't have voice conversion capability
|
||||
raise ValueError("F5-TTS engine does not support voice conversion. Use ChatterBox or CosyVoice engine for voice conversion.")
|
||||
|
||||
@@ -831,14 +842,22 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
)
|
||||
converted_chunk_audio = result[0]
|
||||
|
||||
elif engine_type == "cosyvoice":
|
||||
elif engine_type == "cosyvoice":
|
||||
# CosyVoice VC processor
|
||||
result = engine_instance.convert_voice(
|
||||
source_audio=chunk_audio_dict,
|
||||
target_audio=target_audio,
|
||||
refinement_passes=refinement_passes
|
||||
)
|
||||
converted_chunk_audio = result[0]
|
||||
converted_chunk_audio = result[0]
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
result = engine_instance.convert_voice(
|
||||
source_audio=chunk_audio_dict,
|
||||
target_audio=target_audio,
|
||||
refinement_passes=refinement_passes,
|
||||
)
|
||||
converted_chunk_audio = result[0]
|
||||
|
||||
else:
|
||||
raise ValueError(f"Unsupported engine type for chunking: {engine_type}")
|
||||
@@ -917,8 +936,13 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
print(f"🔄 Voice Changer: Starting {engine_type} voice conversion")
|
||||
|
||||
# Validate engine supports voice conversion
|
||||
if engine_type not in ["chatterbox", "chatterbox_official_23lang", "rvc", "cosyvoice"]:
|
||||
raise ValueError(f"Engine '{engine_type}' does not support voice conversion. Currently supported engines: ChatterBox, ChatterBox Official 23-Lang, RVC, CosyVoice")
|
||||
if engine_type not in ["chatterbox", "chatterbox_official_23lang", "rvc", "cosyvoice", "audio_cpp"]:
|
||||
raise ValueError(f"Engine '{engine_type}' does not support voice conversion. Currently supported engines: ChatterBox, ChatterBox Official 23-Lang, RVC, CosyVoice, audio.cpp")
|
||||
if engine_type == "audio_cpp" and "voice_conversion" not in TTS_engine.get("capabilities", []):
|
||||
family = config.get("family", "selected family")
|
||||
raise ValueError(
|
||||
f"audio.cpp family '{family}' does not map to the Suite's source/target Voice Changer contract"
|
||||
)
|
||||
|
||||
# Extract audio data from flexible inputs (support both AUDIO and NARRATOR_VOICE types)
|
||||
processed_source_audio = self._extract_audio_from_input(source_audio, "source_audio")
|
||||
@@ -1079,7 +1103,7 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
f"Conversion completed successfully"
|
||||
)
|
||||
|
||||
elif engine_type == "cosyvoice":
|
||||
elif engine_type == "cosyvoice":
|
||||
# CosyVoice voice conversion
|
||||
print(f"🔄 Voice Changer: Using CosyVoice3 for voice conversion")
|
||||
|
||||
@@ -1120,12 +1144,45 @@ class UnifiedVoiceChangerNode(BaseVCNode):
|
||||
)
|
||||
|
||||
# Add unified wrapper info
|
||||
conversion_info = (
|
||||
f"🔄 Voice Changer (Unified) - COSYVOICE3 Engine:\n"
|
||||
f"{conversion_info}"
|
||||
)
|
||||
|
||||
else:
|
||||
conversion_info = (
|
||||
f"🔄 Voice Changer (Unified) - COSYVOICE3 Engine:\n"
|
||||
f"{conversion_info}"
|
||||
)
|
||||
|
||||
elif engine_type == "audio_cpp":
|
||||
if len(source_chunks) > 1:
|
||||
converted_waveform, output_sample_rate = self._process_chunks_with_conversion(
|
||||
source_chunks,
|
||||
processed_narrator_target,
|
||||
engine_instance,
|
||||
engine_type,
|
||||
refinement_passes,
|
||||
config,
|
||||
source_sample_rate,
|
||||
)
|
||||
converted_audio = {
|
||||
"waveform": converted_waveform,
|
||||
"sample_rate": output_sample_rate,
|
||||
}
|
||||
conversion_info = (
|
||||
f"Model family: {config.get('family', 'external')}\n"
|
||||
f"Chunks: {len(source_chunks)} ({chunk_method}, {max_chunk_duration}s max)\n"
|
||||
f"Refinement passes: {refinement_passes}\n"
|
||||
f"Output sample rate: {output_sample_rate} Hz\n"
|
||||
"Conversion completed successfully"
|
||||
)
|
||||
else:
|
||||
converted_audio, conversion_info = engine_instance.convert_voice(
|
||||
source_audio=processed_source_audio,
|
||||
target_audio=processed_narrator_target,
|
||||
refinement_passes=refinement_passes,
|
||||
)
|
||||
conversion_info = (
|
||||
"🔄 Voice Changer (Unified) - AUDIO.CPP Engine:\n"
|
||||
f"{conversion_info}"
|
||||
)
|
||||
|
||||
else:
|
||||
# Future engines will be handled here
|
||||
raise ValueError(f"Engine type '{engine_type}' voice conversion not yet implemented")
|
||||
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp.capabilities import (
|
||||
get_package_dependencies,
|
||||
get_capability,
|
||||
load_capabilities,
|
||||
public_capabilities,
|
||||
validate_voice_reference,
|
||||
)
|
||||
from utils.audio_cpp.catalog import load_catalog
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_capability_overlay_covers_the_pinned_catalog():
|
||||
capabilities = load_capabilities()
|
||||
assert set(capabilities) == set(load_catalog().families)
|
||||
assert capabilities["vibevoice"]["native_multi_speaker"] == {
|
||||
"supported": True,
|
||||
"max_speakers": 4,
|
||||
"suite_status": "partial",
|
||||
}
|
||||
assert capabilities["vibevoice_asr"]["asr_features"] == {
|
||||
"diarization": "native",
|
||||
"timing": "native_segment",
|
||||
}
|
||||
assert capabilities["nemotron_asr"]["asr_features"] == {
|
||||
"diarization": "none",
|
||||
"timing": "native_word",
|
||||
}
|
||||
assert capabilities["qwen3_asr"]["asr_features"]["timing"] == "optional_forced_aligner"
|
||||
assert capabilities["voxtral_realtime"]["asr_features"] == {
|
||||
"diarization": "none",
|
||||
"timing": "none",
|
||||
}
|
||||
public = public_capabilities()
|
||||
assert set(public["packages"]) == set(load_catalog().packages)
|
||||
assert public["packages"]["qwen3_tts_1_7b_base_q8_0"]["estimated_download_bytes"] == 2695175104
|
||||
mio = public["packages"]["miotts_1_7b_q8_0"]
|
||||
assert mio["dependencies"] == ["miocodec_q8_0"]
|
||||
assert mio["estimated_download_bytes"] == 2496393216
|
||||
assert get_package_dependencies("miotts_1_7b_q8_0")[0]["session_option"] == "miotts.codec_model_path"
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_glm_requires_audio_and_matching_transcript():
|
||||
with pytest.raises(ValueError, match="requires reference audio"):
|
||||
validate_voice_reference("glm_tts", {}, "Alice")
|
||||
with pytest.raises(ValueError, match="requires the transcript"):
|
||||
validate_voice_reference(
|
||||
"glm_tts",
|
||||
{"audio": {"waveform": object(), "sample_rate": 24000}},
|
||||
"Alice",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_optional_reference_family_accepts_default_voice():
|
||||
validate_voice_reference("pocket_tts", {}, "narrator")
|
||||
assert get_capability("supertonic")["built_in_voices"] is True
|
||||
@@ -0,0 +1,76 @@
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp.catalog import (
|
||||
AUDIO_CPP_RELEASE_VERSION,
|
||||
CatalogError,
|
||||
family_choices,
|
||||
get_model_specs_dir,
|
||||
load_catalog,
|
||||
package_choices,
|
||||
recommended_package,
|
||||
resolve_task,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_pinned_release_catalog_has_exact_suite_compatible_surface():
|
||||
catalog = load_catalog()
|
||||
|
||||
assert AUDIO_CPP_RELEASE_VERSION == "0.5.1"
|
||||
assert len(catalog.families) == 32
|
||||
assert len(catalog.packages) == 96
|
||||
assert set(family_choices()) == set(catalog.families)
|
||||
assert len(package_choices()) == 96
|
||||
assert set(path.name for path in get_model_specs_dir().glob("*.json")) == {
|
||||
family.spec_filename for family in catalog.families.values()
|
||||
}
|
||||
assert "vevo2" in catalog.families
|
||||
assert catalog.family("vevo2").runtime_tasks == ("tts", "vc", "s2s", "svc")
|
||||
assert catalog.family("qwen3_asr").runtime_tasks == ("asr",)
|
||||
assert catalog.family("seed_vc").runtime_tasks == ("vc", "svc")
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_catalog_merges_package_download_defaults_and_maps_local_paths():
|
||||
catalog = load_catalog()
|
||||
package = catalog.package("chatterbox_q8_0")
|
||||
|
||||
assert package.repo == "audio-cpp/audio.cpp-gguf"
|
||||
assert package.revision == "main"
|
||||
assert package.local_files == (Path("chatterbox-q8_0.gguf"),)
|
||||
assert recommended_package("chatterbox") == "chatterbox_q8_0"
|
||||
|
||||
# Upstream release-0.5.1 uses strip_prefix="." here. It means no strip,
|
||||
# not a literal directory named dot.
|
||||
assert catalog.package("vietneu_tts_v3_turbo_q8_0").local_files == (Path("model.gguf"),)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_resolve_task_uses_compiled_ids_and_specialized_package_semantics():
|
||||
assert resolve_task("chatterbox", "chatterbox_q8_0", "clone") == "clon"
|
||||
assert (
|
||||
resolve_task("qwen3_tts", "qwen3_tts_1_7b_voicedesign_q8_0", "auto") == "vdes"
|
||||
)
|
||||
assert (
|
||||
resolve_task("irodori_tts", "irodori_tts_600m_v3_voicedesign_f16", "auto") == "vdes"
|
||||
)
|
||||
assert resolve_task("qwen3_tts", "qwen3_tts_1_7b_base_q8_0", "auto") == "tts"
|
||||
assert resolve_task("pocket_tts", "pocket_tts_english_q8_0", "clone") == "tts"
|
||||
with pytest.raises(CatalogError, match="does not belong"):
|
||||
resolve_task("chatterbox", "vevo2_q8_0", "auto")
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_every_package_maps_to_safe_relative_files():
|
||||
for package in load_catalog().packages.values():
|
||||
assert package.local_files
|
||||
for path in package.local_files:
|
||||
assert not path.is_absolute()
|
||||
assert ".." not in path.parts
|
||||
@@ -0,0 +1,103 @@
|
||||
import io
|
||||
from pathlib import Path
|
||||
import sys
|
||||
import urllib.error
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp.catalog import load_catalog
|
||||
from utils.audio_cpp.downloader import AudioCppDownloadError, install_package, package_download_size
|
||||
from utils.audio_cpp.discovery import package_install_path
|
||||
|
||||
|
||||
class FakeResponse(io.BytesIO):
|
||||
def __init__(self, payload, content_length=None):
|
||||
super().__init__(payload)
|
||||
self.status = 200
|
||||
self.headers = {
|
||||
"Content-Length": str(len(payload) if content_length is None else content_length)
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_package_download_size_uses_hf_metadata_without_downloading():
|
||||
package = load_catalog().package("pocket_tts_english_q8_0")
|
||||
requests = []
|
||||
|
||||
def opener(request, timeout):
|
||||
requests.append(request)
|
||||
return FakeResponse(b"", content_length=123_456)
|
||||
|
||||
assert package_download_size(package, token="secret-token", opener=opener) == 123_456
|
||||
assert requests[0].method == "HEAD"
|
||||
assert requests[0].get_header("Authorization") == "Bearer secret-token"
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_direct_hf_download_uses_auth_staging_and_nested_atomic_publish(tmp_path):
|
||||
package = load_catalog().package("pocket_tts_english_q8_0")
|
||||
payload = b"complete-gguf"
|
||||
requests = []
|
||||
|
||||
def opener(request, timeout):
|
||||
requests.append((request, timeout))
|
||||
return FakeResponse(payload)
|
||||
|
||||
result = install_package(package, tmp_path, token="secret-token", opener=opener)
|
||||
target = package_install_path(package, tmp_path)
|
||||
|
||||
assert result.path == target
|
||||
assert (target / package.local_files[0]).read_bytes() == payload
|
||||
assert requests[0][0].get_header("Authorization") == "Bearer secret-token"
|
||||
assert "huggingface.co/audio-cpp/audio.cpp-gguf/resolve/main/" in requests[0][0].full_url
|
||||
assert not list(target.parent.glob("*.staging"))
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_incomplete_http_response_never_publishes_package(tmp_path):
|
||||
package = load_catalog().package("chatterbox_q8_0")
|
||||
|
||||
def opener(request, timeout):
|
||||
return FakeResponse(b"short", content_length=100)
|
||||
|
||||
with pytest.raises(AudioCppDownloadError, match="Incomplete download"):
|
||||
install_package(package, tmp_path, opener=opener)
|
||||
|
||||
assert not package_install_path(package, tmp_path).exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_install_preserves_sibling_precision_in_shared_target(tmp_path):
|
||||
package = load_catalog().package("chatterbox_q8_0")
|
||||
sibling_package = load_catalog().package("chatterbox_f16")
|
||||
target = package_install_path(package, tmp_path)
|
||||
sibling = package_install_path(sibling_package, tmp_path)
|
||||
target.mkdir(parents=True)
|
||||
sibling_file = sibling / sibling_package.local_files[0]
|
||||
sibling_file.write_bytes(b"keep")
|
||||
|
||||
result = install_package(
|
||||
package,
|
||||
tmp_path,
|
||||
opener=lambda request, timeout: FakeResponse(b"new"),
|
||||
)
|
||||
|
||||
assert (result.path / package.local_files[0]).read_bytes() == b"new"
|
||||
assert sibling_file.read_bytes() == b"keep"
|
||||
assert not list(target.parent.glob("*.backup"))
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_hf_auth_failure_is_actionable_and_leaves_no_target(tmp_path):
|
||||
package = load_catalog().package("chatterbox_q8_0")
|
||||
|
||||
def opener(request, timeout):
|
||||
raise urllib.error.HTTPError(request.full_url, 401, "Unauthorized", {}, None)
|
||||
|
||||
with pytest.raises(AudioCppDownloadError, match="HF_TOKEN"):
|
||||
install_package(package, tmp_path, opener=opener)
|
||||
assert not package_install_path(package, tmp_path).exists()
|
||||
@@ -0,0 +1,346 @@
|
||||
"""Focused tests for the audio.cpp adapter, processors, and engine node."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def _load_module(name, relative_path):
|
||||
spec = importlib.util.spec_from_file_location(name, PROJECT_ROOT / relative_path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
adapter_module = _load_module("audio_cpp_adapter_test_module", "engines/adapters/audio_cpp_adapter.py")
|
||||
processor_module = _load_module("audio_cpp_processor_test_module", "nodes/audio_cpp/audio_cpp_processor.py")
|
||||
srt_module = _load_module("audio_cpp_srt_test_module", "nodes/audio_cpp/audio_cpp_srt_processor.py")
|
||||
node_module = _load_module("audio_cpp_engine_node_test_module", "nodes/engines/audio_cpp_engine_node.py")
|
||||
|
||||
|
||||
class _FakeSession:
|
||||
def __init__(self, sample_rate=32000):
|
||||
self.sample_rate = sample_rate
|
||||
self.requests = []
|
||||
self.owned = True
|
||||
self.endpoint = ""
|
||||
self.model_id = "owned-test-model"
|
||||
self.family = "qwen3_tts"
|
||||
self.config = {"task": "tts"}
|
||||
|
||||
def run(self, request):
|
||||
self.requests.append(dict(request))
|
||||
# Owned sessions receive a random HTTP endpoint only after the server
|
||||
# starts. That transient port must not change the audio cache identity.
|
||||
self.endpoint = "http://127.0.0.1:54321"
|
||||
if request.get("voice_ref"):
|
||||
from pathlib import Path
|
||||
|
||||
assert Path(request["voice_ref"]).is_absolute()
|
||||
assert Path(request["voice_ref"]).is_file()
|
||||
return SimpleNamespace(
|
||||
waveform=torch.ones(1, self.sample_rate // 10),
|
||||
sample_rate=self.sample_rate,
|
||||
named_audio={},
|
||||
)
|
||||
|
||||
|
||||
def test_adapter_caches_real_sample_rate_and_cleans_reference(monkeypatch, tmp_path):
|
||||
adapter_module.get_audio_cache().clear_cache()
|
||||
adapter_module._CACHE_SAMPLE_RATES.clear()
|
||||
session = _FakeSession(sample_rate=32000)
|
||||
monkeypatch.setattr(adapter_module, "_get_session", lambda config: session)
|
||||
created = []
|
||||
|
||||
def fake_save(waveform, sample_rate):
|
||||
path = tmp_path / f"reference-{len(created)}.wav"
|
||||
path.write_bytes(b"temporary")
|
||||
created.append(path)
|
||||
return str(path)
|
||||
|
||||
monkeypatch.setattr(
|
||||
adapter_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
|
||||
)
|
||||
adapter = adapter_module.AudioCppEngineAdapter(
|
||||
{"family": "qwen3_tts", "package_id": "qwen3_tts_1_7b_base_q8_0", "task": "tts"}
|
||||
)
|
||||
voice = {"audio": {"waveform": torch.zeros(1, 80), "sample_rate": 16000}}
|
||||
|
||||
first, first_rate = adapter.generate_single("hello", voice, seed=7)
|
||||
second, second_rate = adapter.generate_single("hello", voice, seed=7)
|
||||
|
||||
assert first_rate == second_rate == 32000
|
||||
assert torch.equal(first, second)
|
||||
assert len(session.requests) == 1
|
||||
assert session.requests[0]["seed"] == "7"
|
||||
assert len(created) == 1
|
||||
assert created[0].exists()
|
||||
adapter.close()
|
||||
assert all(not path.exists() for path in created)
|
||||
|
||||
|
||||
class _ProcessorAdapter:
|
||||
def __init__(self, sample_rate=24000):
|
||||
self.sample_rate = sample_rate
|
||||
self.config = {}
|
||||
|
||||
def update_config(self, config):
|
||||
self.config = dict(config)
|
||||
|
||||
def generate_single(self, **kwargs):
|
||||
return torch.ones(1, self.sample_rate // 10), self.sample_rate
|
||||
|
||||
|
||||
def test_processor_materializes_leading_pause_at_response_rate(monkeypatch):
|
||||
segment = SimpleNamespace(
|
||||
text="[pause:0.01] hello",
|
||||
character="narrator",
|
||||
parameters={},
|
||||
language=None,
|
||||
explicit_language=False,
|
||||
)
|
||||
monkeypatch.setattr(processor_module.AudioCppProcessor, "_setup_character_parser", lambda self, text: None)
|
||||
monkeypatch.setattr(
|
||||
processor_module.character_parser,
|
||||
"parse_text_segments",
|
||||
lambda text, engine_type=None: [segment],
|
||||
)
|
||||
monkeypatch.setattr(processor_module, "get_character_mapping", lambda *args, **kwargs: {})
|
||||
processor = processor_module.AudioCppProcessor(_ProcessorAdapter(32000), {"language": "auto"})
|
||||
|
||||
records = processor.process_text(
|
||||
"ignored", {}, seed=1, enable_chunking=False, show_text_logging=False
|
||||
)
|
||||
|
||||
assert processor.sample_rate == 32000
|
||||
assert records[0]["sample_rate"] == 32000
|
||||
assert records[0]["waveform"].shape == (1, 320)
|
||||
assert records[1]["sample_rate"] == 32000
|
||||
|
||||
|
||||
def test_processor_uses_glm_transcripts_and_resets_rate_between_generations(monkeypatch):
|
||||
segment = SimpleNamespace(
|
||||
text="hello",
|
||||
character="Alice",
|
||||
parameters={},
|
||||
language=None,
|
||||
explicit_language=False,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
processor_module.AudioCppProcessor, "_setup_character_parser", lambda self, text: None
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
processor_module.character_parser,
|
||||
"parse_text_segments",
|
||||
lambda text, engine_type=None: [segment],
|
||||
)
|
||||
discovery_modes = []
|
||||
|
||||
def mapping(characters, engine_type):
|
||||
discovery_modes.append(engine_type)
|
||||
return {"Alice": ("alice.wav", "matching transcript")}
|
||||
|
||||
monkeypatch.setattr(processor_module, "get_character_mapping", mapping)
|
||||
adapter = _ProcessorAdapter(24000)
|
||||
processor = processor_module.AudioCppProcessor(adapter, {"family": "glm_tts"})
|
||||
|
||||
processor.process_text("first", {}, seed=1, enable_chunking=False, show_text_logging=False)
|
||||
adapter.sample_rate = 32000
|
||||
processor.process_text("second", {}, seed=1, enable_chunking=False, show_text_logging=False)
|
||||
|
||||
assert discovery_modes == ["audio_and_text", "audio_and_text"]
|
||||
assert processor.sample_rate == 32000
|
||||
|
||||
|
||||
def test_processor_rejects_mixed_response_rates():
|
||||
processor = processor_module.AudioCppProcessor(_ProcessorAdapter(), {})
|
||||
segments = [
|
||||
{"waveform": torch.zeros(1, 8), "sample_rate": 24000, "text": "a"},
|
||||
{"waveform": torch.zeros(1, 8), "sample_rate": 32000, "text": "b"},
|
||||
]
|
||||
with pytest.raises(RuntimeError, match="inconsistent sample rates"):
|
||||
processor.combine_audio_segments(segments)
|
||||
|
||||
|
||||
def test_engine_node_resolves_owned_package_task(monkeypatch):
|
||||
monkeypatch.setattr(node_module, "_recommended_package", lambda family: "design-package")
|
||||
monkeypatch.setattr(node_module, "_validate_package", lambda family, package: None)
|
||||
monkeypatch.setattr(node_module, "_resolve_task", lambda family, package, task: "vdes")
|
||||
|
||||
engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"managed", "qwen3_tts", "auto", "auto", "cuda", 0, 4, "auto"
|
||||
)[0]
|
||||
|
||||
assert engine["config"]["package_id"] == "design-package"
|
||||
assert engine["config"]["task"] == "vdes"
|
||||
assert engine["config"]["threads"] == 4
|
||||
assert engine["capabilities"] == ["tts", "voice_design"]
|
||||
|
||||
|
||||
def test_engine_node_keeps_external_server_task_authoritative():
|
||||
engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"external_server",
|
||||
"qwen3_tts",
|
||||
"auto",
|
||||
"auto",
|
||||
"cpu",
|
||||
0,
|
||||
4,
|
||||
"auto",
|
||||
server_url="http://127.0.0.1:8080",
|
||||
)[0]
|
||||
assert engine["config"]["task"] == "auto"
|
||||
assert engine["config"]["package_id"] == "auto"
|
||||
|
||||
|
||||
def test_engine_node_uses_pinned_catalog_contract():
|
||||
inputs = node_module.AudioCppEngineNode.INPUT_TYPES()
|
||||
assert "qwen3_tts" in inputs["required"]["family"][0]
|
||||
assert "qwen3_tts_1_7b_base_q8_0" in inputs["required"]["package_id"][0]
|
||||
|
||||
engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"managed", "qwen3_tts", "auto", "auto", "cpu", 0, 4, "auto"
|
||||
)[0]
|
||||
assert engine["config"]["package_id"] == "qwen3_tts_1_7b_base_q8_0"
|
||||
assert engine["config"]["task"] == "tts"
|
||||
|
||||
|
||||
def test_unified_nodes_construct_audio_cpp_processors_without_nodes_package_collision():
|
||||
text_module = _load_module(
|
||||
"audio_cpp_unified_text_test_module", "nodes/unified/tts_text_node.py"
|
||||
)
|
||||
srt_unified_module = _load_module(
|
||||
"audio_cpp_unified_srt_test_module", "nodes/unified/tts_srt_node.py"
|
||||
)
|
||||
engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"external_server",
|
||||
"pocket_tts",
|
||||
"auto",
|
||||
"auto",
|
||||
"cpu",
|
||||
0,
|
||||
4,
|
||||
"auto",
|
||||
server_url="http://127.0.0.1:9999",
|
||||
model_id="wiring-only",
|
||||
)[0]
|
||||
|
||||
text_wrapper = text_module.UnifiedTTSTextNode()._create_proper_engine_node_instance(engine)
|
||||
srt_wrapper = srt_unified_module.UnifiedTTSSRTNode()._create_proper_engine_node_instance(engine)
|
||||
|
||||
assert type(text_wrapper.adapter).__name__ == "AudioCppEngineAdapter"
|
||||
assert type(text_wrapper.processor).__name__ == "AudioCppProcessor"
|
||||
assert type(srt_wrapper.processor).__name__ == "AudioCppSRTProcessor"
|
||||
|
||||
|
||||
def test_unified_nodes_surface_audio_cpp_runtime_errors(monkeypatch):
|
||||
from utils.audio_cpp import session as session_module
|
||||
|
||||
text_module = _load_module(
|
||||
"audio_cpp_unified_text_error_test_module", "nodes/unified/tts_text_node.py"
|
||||
)
|
||||
srt_unified_module = _load_module(
|
||||
"audio_cpp_unified_srt_error_test_module", "nodes/unified/tts_srt_node.py"
|
||||
)
|
||||
engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"external_server",
|
||||
"pocket_tts",
|
||||
"auto",
|
||||
"auto",
|
||||
"cpu",
|
||||
0,
|
||||
4,
|
||||
"auto",
|
||||
server_url="http://127.0.0.1:9999",
|
||||
model_id="error-only",
|
||||
)[0]
|
||||
|
||||
def fail_session(config):
|
||||
raise RuntimeError("visible audio.cpp failure")
|
||||
|
||||
monkeypatch.setattr(session_module, "get_audio_cpp_session", fail_session)
|
||||
|
||||
with pytest.raises(RuntimeError, match="visible audio.cpp failure"):
|
||||
text_module.UnifiedTTSTextNode().generate_speech(
|
||||
engine, "hello", "none", 1, enable_chunking=False, enable_audio_cache=False
|
||||
)
|
||||
with pytest.raises(RuntimeError, match="visible audio.cpp failure"):
|
||||
srt_unified_module.UnifiedTTSSRTNode().generate_srt_speech(
|
||||
engine,
|
||||
"1\n00:00:00,000 --> 00:00:01,000\nhello",
|
||||
"none",
|
||||
1,
|
||||
"concatenate",
|
||||
enable_audio_cache=False,
|
||||
)
|
||||
|
||||
|
||||
class _Subtitle:
|
||||
def __init__(self, sequence, text, start, end):
|
||||
self.sequence = sequence
|
||||
self.text = text
|
||||
self.start_time = start
|
||||
self.end_time = end
|
||||
self.duration = end - start
|
||||
|
||||
|
||||
class _SRTTextProcessor:
|
||||
sample_rate = 16000
|
||||
|
||||
def reset_sample_rate(self):
|
||||
return None
|
||||
|
||||
def process_text(self, **kwargs):
|
||||
return [{"waveform": torch.ones(1, 8000), "sample_rate": 16000, "text": kwargs["text"]}]
|
||||
|
||||
def combine_audio_segments(self, records, **kwargs):
|
||||
return records[0]["waveform"]
|
||||
|
||||
|
||||
def test_srt_delays_blank_cue_until_dynamic_rate_is_known(monkeypatch):
|
||||
subtitles = [_Subtitle(1, "", 0.0, 0.25), _Subtitle(2, "hello", 0.25, 0.75)]
|
||||
instance = srt_module.AudioCppSRTProcessor.__new__(srt_module.AudioCppSRTProcessor)
|
||||
instance.config = {}
|
||||
instance._processor = _SRTTextProcessor()
|
||||
instance.SRTParser = lambda: SimpleNamespace(
|
||||
parse_srt_content=lambda content, allow_overlaps: subtitles
|
||||
)
|
||||
monkeypatch.setattr(instance, "_check_interrupt", lambda *args: None)
|
||||
monkeypatch.setattr(srt_module.SRTOverlapHandler, "detect_overlaps", lambda items: False)
|
||||
monkeypatch.setattr(
|
||||
srt_module.SRTOverlapHandler,
|
||||
"handle_smart_natural_fallback",
|
||||
lambda mode, overlaps, label: (mode, False),
|
||||
)
|
||||
captured = {}
|
||||
|
||||
def fake_assemble(audio, subs, mode, params, rate):
|
||||
captured["segments"] = audio
|
||||
return torch.cat(audio, dim=-1), None, None
|
||||
|
||||
monkeypatch.setattr(instance, "_assemble", fake_assemble)
|
||||
monkeypatch.setattr(
|
||||
srt_module,
|
||||
"SRTReportGenerator",
|
||||
lambda: SimpleNamespace(
|
||||
generate_timing_report=lambda *args: "report",
|
||||
generate_adjusted_srt_string=lambda *args: "adjusted",
|
||||
),
|
||||
)
|
||||
|
||||
audio, _, report, adjusted = instance.process_srt_content(
|
||||
"unused", {}, 0, "concatenate", {}, enable_audio_cache=False
|
||||
)
|
||||
|
||||
assert captured["segments"][0].shape[-1] == 4000
|
||||
assert audio["sample_rate"] == 16000
|
||||
assert audio["waveform"].shape == (1, 1, 12000)
|
||||
assert (report, adjusted) == ("report", "adjusted")
|
||||
@@ -0,0 +1,260 @@
|
||||
"""No-model tests for audio.cpp ASR and unified voice-conversion contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from utils.asr.types import ASRRequest
|
||||
from utils.audio_cpp import session as session_module
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def _load_module(name: str, relative_path: str):
|
||||
spec = importlib.util.spec_from_file_location(name, PROJECT_ROOT / relative_path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
asr_module = _load_module(
|
||||
"audio_cpp_asr_adapter_test_module", "engines/adapters/asr_audio_cpp_adapter.py"
|
||||
)
|
||||
vc_module = _load_module(
|
||||
"audio_cpp_vc_adapter_test_module", "engines/adapters/audio_cpp_vc_adapter.py"
|
||||
)
|
||||
node_module = _load_module(
|
||||
"audio_cpp_multitask_node_test_module", "nodes/engines/audio_cpp_engine_node.py"
|
||||
)
|
||||
|
||||
|
||||
def _fake_save_factory(tmp_path):
|
||||
paths = []
|
||||
|
||||
def save(_waveform, _sample_rate):
|
||||
path = tmp_path / f"audio-{len(paths)}.wav"
|
||||
path.write_bytes(b"wav")
|
||||
paths.append(path)
|
||||
return str(path)
|
||||
|
||||
return paths, save
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_engine_node_advertises_asr_and_vc_consumers():
|
||||
asr_engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"managed", "qwen3_asr", "auto", "auto", "cpu", 0, 4, "auto"
|
||||
)[0]
|
||||
vc_engine = node_module.AudioCppEngineNode().create_engine_config(
|
||||
"managed", "seed_vc", "auto", "auto", "cpu", 0, 4, "auto"
|
||||
)[0]
|
||||
|
||||
assert asr_engine["config"]["task"] == "asr"
|
||||
assert asr_engine["capabilities"] == ["asr"]
|
||||
assert vc_engine["config"]["task"] == "vc"
|
||||
assert vc_engine["capabilities"] == ["voice_conversion"]
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_asr_adapter_normalizes_words_and_speaker_turns(monkeypatch, tmp_path):
|
||||
paths, fake_save = _fake_save_factory(tmp_path)
|
||||
monkeypatch.setattr(
|
||||
asr_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
|
||||
)
|
||||
|
||||
class FakeSession:
|
||||
task = "asr"
|
||||
model_id = "vibe-asr"
|
||||
|
||||
def run(self, request):
|
||||
assert Path(request["audio"]).is_absolute()
|
||||
assert Path(request["audio"]).is_file()
|
||||
return SimpleNamespace(raw={
|
||||
"text": "hello world",
|
||||
"language": "en",
|
||||
"words": [
|
||||
{"word": "hello", "start_sample": 0, "end_sample": 8000},
|
||||
{"word": "world", "start_sample": 8000, "end_sample": 16000},
|
||||
],
|
||||
"speaker_turns": [
|
||||
{
|
||||
"start_sample": 0,
|
||||
"end_sample": 16000,
|
||||
"speaker_id": "Speaker 1",
|
||||
"text": "hello world",
|
||||
}
|
||||
],
|
||||
})
|
||||
|
||||
monkeypatch.setattr(session_module, "get_audio_cpp_session", lambda _config: FakeSession())
|
||||
adapter = asr_module.AudioCppASREngineAdapter({
|
||||
"engine_type": "audio_cpp",
|
||||
"config": {"family": "vibevoice_asr", "connection_mode": "external_server"},
|
||||
})
|
||||
result = adapter.transcribe(ASRRequest(
|
||||
audio={"waveform": torch.zeros(1, 1, 16000), "sample_rate": 16000},
|
||||
timestamps="word",
|
||||
diarization=True,
|
||||
chunk_size=0,
|
||||
))
|
||||
|
||||
assert result.text == "[Speaker 1] hello world"
|
||||
assert result.language == "en"
|
||||
assert result.segments[0].speaker == "Speaker 1"
|
||||
assert [word.text for word in result.segments[0].words] == ["hello", "world"]
|
||||
assert paths and not paths[0].exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_asr_adapter_uses_suite_chunking_and_deduplicates_overlap(monkeypatch, tmp_path):
|
||||
paths, fake_save = _fake_save_factory(tmp_path)
|
||||
monkeypatch.setattr(
|
||||
asr_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
|
||||
)
|
||||
|
||||
class FakeSession:
|
||||
task = "asr"
|
||||
model_id = "nemotron-asr"
|
||||
owned = True
|
||||
|
||||
def __init__(self):
|
||||
self.requests = []
|
||||
self.restarts = 0
|
||||
self.texts = iter((
|
||||
"one two three",
|
||||
"three four five",
|
||||
"five six seven",
|
||||
))
|
||||
|
||||
def restart_owned_runtime(self):
|
||||
self.restarts += 1
|
||||
|
||||
def run(self, request):
|
||||
self.requests.append(request)
|
||||
assert Path(request["audio"]).is_file()
|
||||
return SimpleNamespace(raw={"text": next(self.texts), "language": "en"})
|
||||
|
||||
fake_session = FakeSession()
|
||||
monkeypatch.setattr(
|
||||
session_module, "get_audio_cpp_session", lambda _config: fake_session
|
||||
)
|
||||
adapter = asr_module.AudioCppASREngineAdapter({
|
||||
"engine_type": "audio_cpp",
|
||||
"config": {"family": "nemotron_asr", "connection_mode": "external_server"},
|
||||
})
|
||||
result = adapter.transcribe(ASRRequest(
|
||||
audio={"waveform": torch.zeros(1, 1, 80), "sample_rate": 10},
|
||||
chunk_size=4,
|
||||
overlap=2,
|
||||
))
|
||||
|
||||
assert result.text == "one two three four five six seven"
|
||||
assert len(fake_session.requests) == 3
|
||||
assert fake_session.restarts == 2
|
||||
assert result.raw["timing"]["suite_chunks"] == 3
|
||||
assert any("Suite-side ASR chunking" in note for note in result.raw["notes"])
|
||||
assert [chunk["text"] for chunk in result.raw["chunks"]] == [
|
||||
"one two three",
|
||||
"three four five",
|
||||
"five six seven",
|
||||
]
|
||||
assert [(chunk["start"], chunk["end"]) for chunk in result.raw["chunks"]] == [
|
||||
(0.0, 4.0),
|
||||
(2.0, 6.0),
|
||||
(4.0, 8.0),
|
||||
]
|
||||
assert paths and all(not path.exists() for path in paths)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_vibevoice_diarization_keeps_native_chunking(monkeypatch, tmp_path):
|
||||
paths, fake_save = _fake_save_factory(tmp_path)
|
||||
monkeypatch.setattr(
|
||||
asr_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
|
||||
)
|
||||
captured = {}
|
||||
|
||||
class FakeSession:
|
||||
task = "asr"
|
||||
model_id = "vibe-asr"
|
||||
|
||||
def run(self, request):
|
||||
captured.update(request)
|
||||
return SimpleNamespace(raw={
|
||||
"text": "hello",
|
||||
"speaker_turns": [{
|
||||
"start_sample": 0,
|
||||
"end_sample": 16000,
|
||||
"speaker_id": "1",
|
||||
"text": "hello",
|
||||
}],
|
||||
})
|
||||
|
||||
monkeypatch.setattr(
|
||||
session_module, "get_audio_cpp_session", lambda _config: FakeSession()
|
||||
)
|
||||
adapter = asr_module.AudioCppASREngineAdapter({
|
||||
"engine_type": "audio_cpp",
|
||||
"config": {"family": "vibevoice_asr", "connection_mode": "external_server"},
|
||||
})
|
||||
result = adapter.transcribe(ASRRequest(
|
||||
audio={"waveform": torch.zeros(1, 1, 16000), "sample_rate": 16000},
|
||||
diarization=True,
|
||||
chunk_size=30,
|
||||
overlap=2,
|
||||
))
|
||||
|
||||
assert captured["options"]["audio_chunk_mode"] == "fixed"
|
||||
assert captured["options"]["audio_chunk_seconds"] == 30
|
||||
assert result.text == "[Speaker 1] hello"
|
||||
assert len(paths) == 1 and not paths[0].exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_vc_adapter_forces_vc_task_and_uses_source_target_audio(monkeypatch, tmp_path):
|
||||
paths, fake_save = _fake_save_factory(tmp_path)
|
||||
monkeypatch.setattr(
|
||||
vc_module.AudioProcessingUtils, "save_audio_to_temp_file", staticmethod(fake_save)
|
||||
)
|
||||
captured = {}
|
||||
|
||||
class FakeSession:
|
||||
task = "vc"
|
||||
model_id = "seed-vc"
|
||||
family = "seed_vc"
|
||||
|
||||
def run(self, request):
|
||||
captured.update(request)
|
||||
assert Path(request["audio"]).is_file()
|
||||
assert Path(request["voice_ref"]).is_file()
|
||||
return SimpleNamespace(
|
||||
waveform=torch.ones(1, 2400),
|
||||
sample_rate=24000,
|
||||
named_audio={},
|
||||
)
|
||||
|
||||
def fake_session(config):
|
||||
assert config["requested_task"] == "vc"
|
||||
assert config["task"] == "vc"
|
||||
return FakeSession()
|
||||
|
||||
monkeypatch.setattr(session_module, "get_audio_cpp_session", fake_session)
|
||||
adapter = vc_module.AudioCppVoiceConversionAdapter({
|
||||
"family": "seed_vc",
|
||||
"connection_mode": "managed",
|
||||
})
|
||||
audio = {"waveform": torch.zeros(1, 1, 1600), "sample_rate": 16000}
|
||||
converted, info = adapter.convert_voice(audio, audio)
|
||||
|
||||
assert converted["waveform"].shape == (1, 1, 2400)
|
||||
assert converted["sample_rate"] == 24000
|
||||
assert captured["source_audio"] == captured["audio"]
|
||||
assert captured["target_voice"] == captured["voice_ref"]
|
||||
assert "Seed-VC" not in info or "seed_vc" in info
|
||||
assert paths and all(not path.exists() for path in paths)
|
||||
@@ -0,0 +1,332 @@
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp import resolver
|
||||
from utils.audio_cpp.catalog import load_catalog
|
||||
from utils.audio_cpp.discovery import package_install_path
|
||||
from utils.audio_cpp.runtime_installer import runtime_install_path
|
||||
from utils.audio_cpp.settings import AudioCppSettings
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_auto_reuses_machine_external_server_without_loading_owned_dependencies(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"load_settings",
|
||||
lambda: AudioCppSettings(
|
||||
connection_mode="external",
|
||||
external_server_url="HTTP://127.0.0.1:18080/",
|
||||
executable_path="C:/ignored/audiocpp_server.exe",
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"_catalog_module",
|
||||
lambda: (_ for _ in ()).throw(AssertionError("external mode loaded the catalog")),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"_downloader_module",
|
||||
lambda: (_ for _ in ()).throw(AssertionError("external mode loaded a downloader")),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"_runtime_installer_module",
|
||||
lambda: (_ for _ in ()).throw(AssertionError("external mode loaded an installer")),
|
||||
)
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"config": {
|
||||
"connection_mode": "auto",
|
||||
"family": "qwen3_tts",
|
||||
"package_id": "auto",
|
||||
"task": "auto",
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
assert result["connection_mode"] == "external_server"
|
||||
assert result["server_url"] == "http://127.0.0.1:18080"
|
||||
assert result["binary_path"] == ""
|
||||
assert result["model_path"] == ""
|
||||
assert "model_id" not in result
|
||||
assert result["task"] == "auto"
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_owned_explicit_paths_resolve_recommended_package_task_and_cuda(monkeypatch, tmp_path):
|
||||
binary = tmp_path / "audiocpp_server.exe"
|
||||
binary.write_bytes(b"exe")
|
||||
model = tmp_path / "Qwen-VoiceDesign"
|
||||
model.mkdir()
|
||||
monkeypatch.setattr(resolver, "load_settings", lambda: AudioCppSettings())
|
||||
monkeypatch.setattr(resolver, "_cuda_available", lambda: True)
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "managed",
|
||||
"family": "qwen3_tts",
|
||||
"package_id": "qwen3_tts_1_7b_voicedesign_q8_0",
|
||||
"requested_task": "auto",
|
||||
"backend": "auto",
|
||||
"device": "cuda:2",
|
||||
"binary_path": str(binary),
|
||||
"model_path": str(model),
|
||||
"model_id": "My Qwen model",
|
||||
}
|
||||
)
|
||||
|
||||
assert result["connection_mode"] == "owned_process"
|
||||
assert result["package_id"] == "qwen3_tts_1_7b_voicedesign_q8_0"
|
||||
assert result["task"] == "vdes"
|
||||
assert result["backend"] == "cuda"
|
||||
assert result["device_index"] == 2
|
||||
assert result["binary_path"] == str(binary.resolve())
|
||||
assert result["model_path"] == str(model.resolve())
|
||||
assert result["model_id"] == "My-Qwen-model"
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_owned_reuses_external_model_root_and_configured_runtime_root(monkeypatch, tmp_path):
|
||||
external_models = tmp_path / "existing-audio-cpp" / "models"
|
||||
managed_models = tmp_path / "suite" / "audio.cpp" / "models"
|
||||
configured_runtime = tmp_path / "existing-audio-cpp" / "runtime"
|
||||
settings = AudioCppSettings(
|
||||
connection_mode="managed",
|
||||
model_roots=(str(external_models),),
|
||||
managed_model_root=str(managed_models),
|
||||
runtime_root=str(configured_runtime),
|
||||
runtime_backend="cpu",
|
||||
)
|
||||
package = load_catalog().package("chatterbox_q8_0")
|
||||
installed_model = package_install_path(package, external_models)
|
||||
installed_model.mkdir(parents=True)
|
||||
(installed_model / package.local_files[0]).write_bytes(b"gguf")
|
||||
installed_binary = runtime_install_path(configured_runtime, "cpu") / "audiocpp_server.exe"
|
||||
installed_binary.parent.mkdir(parents=True)
|
||||
installed_binary.write_bytes(b"exe")
|
||||
monkeypatch.setattr(resolver, "load_settings", lambda: settings)
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "auto",
|
||||
"family": "chatterbox",
|
||||
"package_id": "auto",
|
||||
"task": "auto",
|
||||
"backend": "auto",
|
||||
"auto_download_model": False,
|
||||
"auto_download_runtime": False,
|
||||
}
|
||||
)
|
||||
|
||||
assert result["package_id"] == "chatterbox_q8_0"
|
||||
assert result["task"] == "clon"
|
||||
assert result["backend"] == "cpu"
|
||||
assert result["model_path"] == str(installed_model.resolve())
|
||||
assert result["binary_path"] == str(installed_binary.resolve())
|
||||
assert not managed_models.exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_missing_assets_download_only_to_managed_roots(monkeypatch, tmp_path):
|
||||
external_models = tmp_path / "external" / "models"
|
||||
managed_models = tmp_path / "managed" / "audio.cpp" / "models"
|
||||
settings = AudioCppSettings(
|
||||
connection_mode="managed",
|
||||
model_roots=(str(external_models),),
|
||||
managed_model_root=str(managed_models),
|
||||
runtime_backend="cpu",
|
||||
)
|
||||
calls = {}
|
||||
real_downloader = resolver._downloader_module()
|
||||
real_runtime = resolver._runtime_installer_module()
|
||||
|
||||
def install_package(package, root, catalog, progress=None):
|
||||
calls["model_root"] = Path(root)
|
||||
calls["model_progress"] = progress
|
||||
target = package_install_path(package, root)
|
||||
target.mkdir(parents=True)
|
||||
(target / package.local_files[0]).write_bytes(b"gguf")
|
||||
return SimpleNamespace(path=target, bytes_downloaded=4)
|
||||
|
||||
def install_runtime(root, backend, progress=None):
|
||||
calls["runtime_root"] = Path(root)
|
||||
calls["backend"] = backend
|
||||
calls["runtime_progress"] = progress
|
||||
executable = runtime_install_path(root, backend) / "audiocpp_server.exe"
|
||||
executable.parent.mkdir(parents=True)
|
||||
executable.write_bytes(b"exe")
|
||||
return SimpleNamespace(executable=executable)
|
||||
|
||||
monkeypatch.setattr(resolver, "load_settings", lambda: settings)
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"_downloader_module",
|
||||
lambda: SimpleNamespace(install_package=install_package),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"_runtime_installer_module",
|
||||
lambda: SimpleNamespace(
|
||||
runtime_install_path=real_runtime.runtime_install_path,
|
||||
get_runtime_manifest=real_runtime.get_runtime_manifest,
|
||||
install_windows_runtime=install_runtime,
|
||||
),
|
||||
)
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "managed",
|
||||
"family": "chatterbox",
|
||||
"package_id": "chatterbox_q8_0",
|
||||
"backend": "cpu",
|
||||
"auto_download_model": True,
|
||||
"auto_download_runtime": True,
|
||||
}
|
||||
)
|
||||
|
||||
assert calls["model_root"] == managed_models
|
||||
assert calls["runtime_root"] == managed_models.parent / "runtime"
|
||||
assert calls["backend"] == "cpu"
|
||||
assert callable(calls["model_progress"])
|
||||
assert callable(calls["runtime_progress"])
|
||||
assert not external_models.exists()
|
||||
assert result["model_path"].startswith(str(managed_models.resolve()))
|
||||
assert result["binary_path"].startswith(str((managed_models.parent / "runtime").resolve()))
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_missing_model_without_permission_has_actionable_error(monkeypatch, tmp_path):
|
||||
managed_models = tmp_path / "managed" / "models"
|
||||
binary = tmp_path / "audiocpp_server.exe"
|
||||
binary.write_bytes(b"exe")
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"load_settings",
|
||||
lambda: AudioCppSettings(managed_model_root=str(managed_models)),
|
||||
)
|
||||
|
||||
with pytest.raises(resolver.AudioCppResolutionError, match="enable auto_download_model"):
|
||||
resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "managed",
|
||||
"family": "chatterbox",
|
||||
"package_id": "chatterbox_q8_0",
|
||||
"backend": "cpu",
|
||||
"binary_path": str(binary),
|
||||
"auto_download_model": False,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_miotts_installs_codec_dependency_and_sets_absolute_session_path(monkeypatch, tmp_path):
|
||||
managed_models = tmp_path / "managed" / "models"
|
||||
catalog = resolver._catalog_module().load_catalog()
|
||||
miotts = catalog.package("miotts_1_7b_q8_0")
|
||||
miotts_dir = package_install_path(miotts, managed_models)
|
||||
for relative in miotts.local_files:
|
||||
target = miotts_dir / relative
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_bytes(b"miotts")
|
||||
binary = tmp_path / "audiocpp_server.exe"
|
||||
binary.write_bytes(b"exe")
|
||||
installed = {}
|
||||
|
||||
def install_package(package, root, **_kwargs):
|
||||
target = package_install_path(package, root)
|
||||
for relative in package.local_files:
|
||||
path = target / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(b"codec")
|
||||
installed[package.id] = target
|
||||
return SimpleNamespace(path=target, bytes_downloaded=5)
|
||||
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"load_settings",
|
||||
lambda: AudioCppSettings(managed_model_root=str(managed_models)),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"_downloader_module",
|
||||
lambda: SimpleNamespace(install_package=install_package),
|
||||
)
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "managed",
|
||||
"family": "miotts",
|
||||
"package_id": "miotts_1_7b_q8_0",
|
||||
"backend": "cpu",
|
||||
"binary_path": str(binary),
|
||||
"auto_download_model": True,
|
||||
}
|
||||
)
|
||||
|
||||
assert "miocodec_q8_0" in installed
|
||||
assert result["session_options"]["miotts.codec_model_path"] == str(
|
||||
installed["miocodec_q8_0"].resolve()
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_owned_existing_binary_allows_explicit_hip_backend(monkeypatch, tmp_path):
|
||||
binary = tmp_path / "audiocpp_server.exe"
|
||||
binary.write_bytes(b"exe")
|
||||
model = tmp_path / "model"
|
||||
model.mkdir()
|
||||
monkeypatch.setattr(resolver, "load_settings", lambda: AudioCppSettings())
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "existing_binary",
|
||||
"family": "chatterbox",
|
||||
"package_id": "chatterbox_q8_0",
|
||||
"backend": "hip",
|
||||
"binary_path": str(binary),
|
||||
"model_path": str(model),
|
||||
}
|
||||
)
|
||||
|
||||
assert result["backend"] == "hip"
|
||||
assert result["binary_path"] == str(binary.resolve())
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_cpu_backend_reuses_installed_cuda_profile_before_downloading_duplicate(
|
||||
monkeypatch, tmp_path
|
||||
):
|
||||
model = tmp_path / "model"
|
||||
model.mkdir()
|
||||
runtime_root = tmp_path / "runtime"
|
||||
cuda_binary = runtime_install_path(runtime_root, "cuda") / "audiocpp_server.exe"
|
||||
cuda_binary.parent.mkdir(parents=True)
|
||||
cuda_binary.write_bytes(b"cuda-exe")
|
||||
monkeypatch.setattr(
|
||||
resolver,
|
||||
"load_settings",
|
||||
lambda: AudioCppSettings(runtime_root=str(runtime_root), runtime_backend="cpu"),
|
||||
)
|
||||
|
||||
result = resolver.resolve_audio_cpp_config(
|
||||
{
|
||||
"connection_mode": "managed",
|
||||
"family": "chatterbox",
|
||||
"package_id": "chatterbox_q8_0",
|
||||
"model_path": str(model),
|
||||
"backend": "auto",
|
||||
"auto_download_runtime": False,
|
||||
}
|
||||
)
|
||||
|
||||
assert result["backend"] == "cpu"
|
||||
assert result["binary_path"] == str(cuda_binary.resolve())
|
||||
@@ -0,0 +1,145 @@
|
||||
import hashlib
|
||||
import io
|
||||
from pathlib import Path
|
||||
import sys
|
||||
import zipfile
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp.runtime_installer import (
|
||||
RuntimeAsset,
|
||||
RuntimeInstallError,
|
||||
RuntimeManifest,
|
||||
get_runtime_manifest,
|
||||
install_windows_runtime,
|
||||
runtime_install_path,
|
||||
)
|
||||
|
||||
|
||||
class FakeResponse(io.BytesIO):
|
||||
def __init__(self, payload):
|
||||
super().__init__(payload)
|
||||
self.status = 200
|
||||
self.headers = {"Content-Length": str(len(payload))}
|
||||
|
||||
|
||||
def make_zip(files):
|
||||
stream = io.BytesIO()
|
||||
with zipfile.ZipFile(stream, "w") as bundle:
|
||||
for name, payload in files.items():
|
||||
bundle.writestr(name, payload)
|
||||
return stream.getvalue()
|
||||
|
||||
|
||||
def fake_manifest(backend, payloads, required):
|
||||
assets = tuple(
|
||||
RuntimeAsset(
|
||||
filename=name,
|
||||
url=f"https://example.test/{name}",
|
||||
size=len(payload),
|
||||
sha256=hashlib.sha256(payload).hexdigest(),
|
||||
)
|
||||
for name, payload in payloads.items()
|
||||
)
|
||||
return RuntimeManifest(backend=backend, assets=assets, required_files=tuple(required))
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_pinned_windows_manifests_include_verified_cuda_runtime_asset():
|
||||
cpu = get_runtime_manifest("cpu")
|
||||
cuda = get_runtime_manifest("cuda")
|
||||
|
||||
assert len(cpu.assets) == 1
|
||||
assert len(cuda.assets) == 2
|
||||
assert cuda.assets[1].filename == "audiocpp-windows-cuda-runtime.zip"
|
||||
assert cuda.assets[1].sha256 == (
|
||||
"46016655aff8f050806d81efd0fe256c15b86527935bfb3896208d4cac6b5ff8"
|
||||
)
|
||||
assert "cublas64_13.dll" in cuda.required_files
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_cuda_runtime_installs_two_verified_archives_atomically(tmp_path):
|
||||
executable_zip = make_zip(
|
||||
{"audiocpp_server.exe": b"server", "audiocpp_cli.exe": b"cli"}
|
||||
)
|
||||
cuda_zip = make_zip({"cublas64_13.dll": b"cublas", "cufft64_12.dll": b"cufft"})
|
||||
payloads = {"runtime.zip": executable_zip, "cuda.zip": cuda_zip}
|
||||
manifest = fake_manifest(
|
||||
"cuda",
|
||||
payloads,
|
||||
("audiocpp_server.exe", "audiocpp_cli.exe", "cublas64_13.dll", "cufft64_12.dll"),
|
||||
)
|
||||
|
||||
def opener(request, timeout):
|
||||
return FakeResponse(payloads[Path(request.full_url).name])
|
||||
|
||||
result = install_windows_runtime(
|
||||
tmp_path,
|
||||
"cuda",
|
||||
manifest=manifest,
|
||||
platform_name="win32",
|
||||
opener=opener,
|
||||
)
|
||||
|
||||
assert result.path == runtime_install_path(tmp_path, "cuda")
|
||||
assert result.executable.read_bytes() == b"server"
|
||||
assert (result.path / "cublas64_13.dll").read_bytes() == b"cublas"
|
||||
assert not list(result.path.parent.glob("*.staging"))
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_runtime_hash_failure_does_not_publish_or_destroy_existing_target(tmp_path):
|
||||
archive = make_zip({"audiocpp_server.exe": b"new", "audiocpp_cli.exe": b"cli"})
|
||||
asset = RuntimeAsset(
|
||||
filename="runtime.zip",
|
||||
url="https://example.test/runtime.zip",
|
||||
size=len(archive),
|
||||
sha256="0" * 64,
|
||||
)
|
||||
manifest = RuntimeManifest(
|
||||
backend="cpu",
|
||||
assets=(asset,),
|
||||
required_files=("audiocpp_server.exe", "audiocpp_cli.exe"),
|
||||
)
|
||||
target = runtime_install_path(tmp_path, "cpu")
|
||||
target.mkdir(parents=True)
|
||||
marker = target / "old.txt"
|
||||
marker.write_bytes(b"old")
|
||||
|
||||
with pytest.raises(RuntimeInstallError, match="SHA256 mismatch"):
|
||||
install_windows_runtime(
|
||||
tmp_path,
|
||||
"cpu",
|
||||
overwrite=True,
|
||||
manifest=manifest,
|
||||
platform_name="win32",
|
||||
opener=lambda request, timeout: FakeResponse(archive),
|
||||
)
|
||||
|
||||
assert marker.read_bytes() == b"old"
|
||||
assert not (target / "audiocpp_server.exe").exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_runtime_rejects_archive_path_traversal(tmp_path):
|
||||
archive = make_zip(
|
||||
{"../escape.dll": b"bad", "audiocpp_server.exe": b"server", "audiocpp_cli.exe": b"cli"}
|
||||
)
|
||||
manifest = fake_manifest(
|
||||
"cpu", {"runtime.zip": archive}, ("audiocpp_server.exe", "audiocpp_cli.exe")
|
||||
)
|
||||
|
||||
with pytest.raises(RuntimeInstallError, match="Unsafe path"):
|
||||
install_windows_runtime(
|
||||
tmp_path,
|
||||
"cpu",
|
||||
manifest=manifest,
|
||||
platform_name="win32",
|
||||
opener=lambda request, timeout: FakeResponse(archive),
|
||||
)
|
||||
assert not (tmp_path / "escape.dll").exists()
|
||||
@@ -0,0 +1,109 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp.catalog import load_catalog
|
||||
from utils.audio_cpp.discovery import (
|
||||
find_installed_package,
|
||||
package_install_path,
|
||||
resolve_model,
|
||||
resolve_model_roots,
|
||||
)
|
||||
from utils.audio_cpp.settings import AudioCppSettings, get_settings_path, load_settings, save_settings
|
||||
|
||||
|
||||
class FakeFolderPaths:
|
||||
def __init__(self, user_root, models_root, registry):
|
||||
self.user_root = Path(user_root)
|
||||
self.models_dir = str(models_root)
|
||||
self.folder_names_and_paths = {
|
||||
key: ([str(path) for path in paths], set()) for key, paths in registry.items()
|
||||
}
|
||||
|
||||
def get_system_user_directory(self, name):
|
||||
return str(self.user_root / name)
|
||||
|
||||
def get_folder_paths(self, name):
|
||||
return list(self.folder_names_and_paths[name][0])
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_settings_live_under_comfyui_system_user_directory_and_round_trip(tmp_path):
|
||||
fake = FakeFolderPaths(tmp_path / "user", tmp_path / "models", {})
|
||||
expected = tmp_path / "user" / "tts_audio_suite" / "audio_cpp" / "settings.json"
|
||||
settings = AudioCppSettings(
|
||||
connection_mode="external",
|
||||
external_server_url="http://127.0.0.1:19090",
|
||||
model_roots=(str(tmp_path / "shared"),),
|
||||
runtime_backend="cuda",
|
||||
extras={"future_key": {"kept": True}},
|
||||
)
|
||||
|
||||
assert get_settings_path(fake) == expected
|
||||
assert save_settings(settings, folder_paths_module=fake) == expected
|
||||
loaded = load_settings(folder_paths_module=fake, strict=True)
|
||||
assert loaded == settings
|
||||
assert json.loads(expected.read_text(encoding="utf-8"))["future_key"] == {"kept": True}
|
||||
assert not list(expected.parent.glob("*.tmp"))
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_broken_settings_fail_safe_unless_strict(tmp_path):
|
||||
path = tmp_path / "settings.json"
|
||||
path.write_text("{broken", encoding="utf-8")
|
||||
|
||||
assert load_settings(path) == AudioCppSettings()
|
||||
with pytest.raises(json.JSONDecodeError):
|
||||
load_settings(path, strict=True)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_model_root_precedence_deduplicates_and_keeps_managed_last(tmp_path):
|
||||
explicit = tmp_path / "explicit"
|
||||
configured = tmp_path / "configured"
|
||||
dedicated = tmp_path / "dedicated"
|
||||
tts_primary = tmp_path / "tts-primary"
|
||||
tts_secondary = tmp_path / "tts-secondary"
|
||||
managed = tts_primary / "audio.cpp" / "models"
|
||||
fake = FakeFolderPaths(
|
||||
tmp_path / "user",
|
||||
tmp_path / "models",
|
||||
{"audio_cpp": [dedicated], "TTS": [tts_primary, tts_secondary]},
|
||||
)
|
||||
settings = AudioCppSettings(
|
||||
model_roots=(str(configured), str(dedicated)),
|
||||
managed_model_root=str(managed),
|
||||
)
|
||||
|
||||
assert resolve_model_roots(
|
||||
[explicit, managed], settings=settings, folder_paths_module=fake
|
||||
) == [
|
||||
explicit,
|
||||
configured,
|
||||
dedicated,
|
||||
tts_secondary / "audio.cpp" / "models",
|
||||
managed,
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_discovery_prefers_existing_external_model_without_copying(tmp_path):
|
||||
package = load_catalog().package("chatterbox_q8_0")
|
||||
external = tmp_path / "existing-audio-cpp-models"
|
||||
managed = tmp_path / "managed"
|
||||
installed = package_install_path(package, external)
|
||||
installed.mkdir(parents=True)
|
||||
(installed / package.local_files[0]).write_bytes(b"gguf")
|
||||
|
||||
assert find_installed_package(package, [external, managed]) == installed
|
||||
resolved = resolve_model(package.id, [external, managed])
|
||||
assert resolved is not None
|
||||
assert resolved.root == external
|
||||
assert resolved.path == installed
|
||||
assert not managed.exists()
|
||||
@@ -0,0 +1,504 @@
|
||||
import base64
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
import threading
|
||||
import types
|
||||
import urllib.request
|
||||
import wave
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
from urllib.parse import parse_qs, urlsplit
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from utils.audio_cpp.client import (
|
||||
AudioCppClient,
|
||||
AudioCppHTTPError,
|
||||
AudioCppProtocolError,
|
||||
)
|
||||
from utils.audio_cpp.catalog import load_catalog
|
||||
from utils.audio_cpp.discovery import package_install_path
|
||||
from utils.audio_cpp.process import AudioCppServerProcess, normalize_audio_cpp_task
|
||||
from utils.audio_cpp.settings import AudioCppSettings
|
||||
from utils.audio_cpp import resolver as audio_cpp_resolver
|
||||
from utils.audio_cpp.session import (
|
||||
audio_cpp_session_statuses,
|
||||
close_all_audio_cpp_sessions,
|
||||
get_audio_cpp_session,
|
||||
)
|
||||
|
||||
|
||||
def _wav_bytes(sample_rate=16000, channels=1, frames=32):
|
||||
buffer = io.BytesIO()
|
||||
with wave.open(buffer, "wb") as wav_file:
|
||||
wav_file.setnchannels(channels)
|
||||
wav_file.setsampwidth(2)
|
||||
wav_file.setframerate(sample_rate)
|
||||
samples = []
|
||||
for index in range(frames):
|
||||
value = int(16000 * ((index % 4) - 1.5) / 1.5)
|
||||
samples.extend([value] * channels)
|
||||
wav_file.writeframes(struct.pack(f"<{len(samples)}h", *samples))
|
||||
return buffer.getvalue()
|
||||
|
||||
|
||||
class _FakeAudioCppHandler(BaseHTTPRequestHandler):
|
||||
server_version = "FakeAudioCpp/0.5.1"
|
||||
|
||||
def log_message(self, format, *args):
|
||||
return None
|
||||
|
||||
def _json(self, payload, status=200):
|
||||
body = json.dumps(payload).encode("utf-8")
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def do_GET(self):
|
||||
parsed = urlsplit(self.path)
|
||||
if parsed.path == "/health":
|
||||
self._json({"status": "ok", "models": 1, "features": ["unload_models"]})
|
||||
elif parsed.path == "/v1/models":
|
||||
self.server.model_queries += 1
|
||||
self._json({
|
||||
"object": "list",
|
||||
"data": [{
|
||||
"id": self.server.model_id,
|
||||
"object": "model",
|
||||
"family": "pocket_tts",
|
||||
"task": "tts",
|
||||
"mode": "offline",
|
||||
}],
|
||||
})
|
||||
elif parsed.path == "/v1/audio/voices":
|
||||
self.server.last_voice_query = parse_qs(parsed.query)
|
||||
self._json({"voices": ["alba", "cosette"]})
|
||||
else:
|
||||
self._json({"error": {"message": "not found", "type": "not_found"}}, 404)
|
||||
|
||||
def do_POST(self):
|
||||
length = int(self.headers.get("Content-Length", "0"))
|
||||
payload = json.loads(self.rfile.read(length).decode("utf-8"))
|
||||
self.server.requests.append(payload)
|
||||
request = payload.get("request", {})
|
||||
text = request.get("text")
|
||||
if text == "http-error":
|
||||
self._json(
|
||||
{"error": {"message": "model is busy", "type": "server_busy"}},
|
||||
503,
|
||||
)
|
||||
return
|
||||
|
||||
encoded = base64.b64encode(self.server.wav_bytes).decode("ascii")
|
||||
if text == "named-only":
|
||||
self._json({
|
||||
"named_audio_outputs": [{
|
||||
"id": "speech",
|
||||
"audio": encoded,
|
||||
"sample_rate": 16000,
|
||||
"channels": 1,
|
||||
}],
|
||||
"timing": {},
|
||||
})
|
||||
elif text == "ambiguous":
|
||||
self._json({
|
||||
"named_audio_outputs": [
|
||||
{"id": "left", "audio": encoded},
|
||||
{"id": "right", "audio": encoded},
|
||||
]
|
||||
})
|
||||
elif text == "transcript-only":
|
||||
self._json({
|
||||
"text": "hello world",
|
||||
"language": "en",
|
||||
"words": [
|
||||
{"word": "hello", "start_sample": 0, "end_sample": 8000},
|
||||
{"word": "world", "start_sample": 8000, "end_sample": 16000},
|
||||
],
|
||||
"timing": {"wall_ms": 1.0},
|
||||
})
|
||||
else:
|
||||
self._json({
|
||||
"audio": encoded,
|
||||
"sample_rate": 16000,
|
||||
"channels": 1,
|
||||
"timing": {"wall_ms": 1.0},
|
||||
})
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def fake_audio_cpp_server():
|
||||
server = ThreadingHTTPServer(("127.0.0.1", 0), _FakeAudioCppHandler)
|
||||
server.model_id = "pocket"
|
||||
server.wav_bytes = _wav_bytes()
|
||||
server.requests = []
|
||||
server.last_voice_query = None
|
||||
server.model_queries = 0
|
||||
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
||||
thread.start()
|
||||
try:
|
||||
yield server, f"http://127.0.0.1:{server.server_port}"
|
||||
finally:
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
thread.join(timeout=2)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def clean_audio_cpp_sessions():
|
||||
close_all_audio_cpp_sessions()
|
||||
yield
|
||||
close_all_audio_cpp_sessions()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_client_health_models_voices_and_primary_audio(fake_audio_cpp_server):
|
||||
server, url = fake_audio_cpp_server
|
||||
client = AudioCppClient(url)
|
||||
|
||||
assert client.health()["status"] == "ok"
|
||||
assert client.models()[0]["id"] == "pocket"
|
||||
assert client.voices("pocket") == ["alba", "cosette"]
|
||||
assert server.last_voice_query == {"model": ["pocket"]}
|
||||
assert client.supports_feature("unload_models") is True
|
||||
|
||||
result = client.run_task("pocket", {"text": "hello", "seed": "42"})
|
||||
assert result.sample_rate == 16000
|
||||
assert result.channels == 1
|
||||
assert result.waveform.shape == (1, 32)
|
||||
assert result.waveform.dtype == torch.float32
|
||||
assert result.waveform.device.type == "cpu"
|
||||
assert server.requests[-1] == {
|
||||
"model": "pocket",
|
||||
"request": {"text": "hello", "seed": "42"},
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_client_selects_sole_named_audio_and_rejects_ambiguous_output(fake_audio_cpp_server):
|
||||
_, url = fake_audio_cpp_server
|
||||
client = AudioCppClient(url)
|
||||
|
||||
result = client.run_task("pocket", {"text": "named-only"})
|
||||
assert result.sample_rate == 16000
|
||||
assert list(result.named_audio) == ["speech"]
|
||||
assert result.waveform.data_ptr() == result.named_audio["speech"].waveform.data_ptr()
|
||||
|
||||
with pytest.raises(AudioCppProtocolError, match="multiple named audio"):
|
||||
client.run_task("pocket", {"text": "ambiguous"})
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_client_accepts_structured_transcript_without_audio(fake_audio_cpp_server):
|
||||
_, url = fake_audio_cpp_server
|
||||
result = AudioCppClient(url).run_task("asr", {"text": "transcript-only"})
|
||||
|
||||
assert result.waveform is None
|
||||
assert result.sample_rate is None
|
||||
assert result.raw["text"] == "hello world"
|
||||
assert len(result.raw["words"]) == 2
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_client_surfaces_structured_http_error(fake_audio_cpp_server):
|
||||
_, url = fake_audio_cpp_server
|
||||
client = AudioCppClient(url)
|
||||
|
||||
with pytest.raises(AudioCppHTTPError) as captured:
|
||||
client.run_task("pocket", {"text": "http-error"})
|
||||
|
||||
assert captured.value.status == 503
|
||||
assert captured.value.error_type == "server_busy"
|
||||
assert "model is busy" in str(captured.value)
|
||||
|
||||
|
||||
def _write_fake_server_script(path: Path) -> Path:
|
||||
script = r'''
|
||||
import argparse
|
||||
import base64
|
||||
import io
|
||||
import json
|
||||
import struct
|
||||
import wave
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--config", required=True)
|
||||
args = parser.parse_args()
|
||||
with open(args.config, "r", encoding="utf-8") as handle:
|
||||
config = json.load(handle)
|
||||
model_id = config["models"][0]["id"]
|
||||
|
||||
buffer = io.BytesIO()
|
||||
with wave.open(buffer, "wb") as wav_file:
|
||||
wav_file.setnchannels(1)
|
||||
wav_file.setsampwidth(2)
|
||||
wav_file.setframerate(22050)
|
||||
wav_file.writeframes(struct.pack("<16h", *range(16)))
|
||||
encoded = base64.b64encode(buffer.getvalue()).decode("ascii")
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
def log_message(self, format, *args):
|
||||
return None
|
||||
def send_json(self, payload):
|
||||
body = json.dumps(payload).encode("utf-8")
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
def do_GET(self):
|
||||
path = urlsplit(self.path).path
|
||||
if path == "/health":
|
||||
self.send_json({"status": "ok", "models": 1})
|
||||
elif path == "/v1/models":
|
||||
self.send_json({"object": "list", "data": [{"id": model_id}]})
|
||||
elif path == "/v1/audio/voices":
|
||||
self.send_json({"voices": ["managed"]})
|
||||
else:
|
||||
self.send_json({})
|
||||
def do_POST(self):
|
||||
length = int(self.headers.get("Content-Length", "0"))
|
||||
json.loads(self.rfile.read(length).decode("utf-8"))
|
||||
self.send_json({"audio": encoded, "sample_rate": 22050, "channels": 1})
|
||||
|
||||
server = ThreadingHTTPServer((config["host"], config["port"]), Handler)
|
||||
print("fake audio.cpp ready", flush=True)
|
||||
server.serve_forever()
|
||||
'''
|
||||
path.write_text(script, encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
def _owned_config(tmp_path: Path, script: Path):
|
||||
model_path = tmp_path / "model"
|
||||
model_path.mkdir(exist_ok=True)
|
||||
(model_path / "weights.gguf").write_bytes(b"audio-cpp-test-model")
|
||||
return {
|
||||
"connection_mode": "existing_binary",
|
||||
"binary_path": sys.executable,
|
||||
"binary_args": [str(script)],
|
||||
"model_path": str(model_path),
|
||||
"model_id": "managed-pocket",
|
||||
"family": "pocket_tts",
|
||||
"package_id": "pocket_tts_english_q8_0",
|
||||
"task": "tts",
|
||||
"backend": "cpu",
|
||||
"startup_timeout": 5.0,
|
||||
"connect_timeout": 1.0,
|
||||
"request_timeout": 5.0,
|
||||
"stop_timeout": 2.0,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_owned_process_writes_secure_config_and_stops_exact_child(tmp_path):
|
||||
assert normalize_audio_cpp_task("clone") == "clon"
|
||||
assert normalize_audio_cpp_task("voice design") == "vdes"
|
||||
script = _write_fake_server_script(tmp_path / "fake_audio_cpp_server.py")
|
||||
runtime = AudioCppServerProcess(_owned_config(tmp_path, script)).start()
|
||||
process = runtime.process
|
||||
config_path = runtime.config_path
|
||||
assert process is not None and process.poll() is None
|
||||
assert config_path is not None and config_path.exists()
|
||||
|
||||
payload = json.loads(config_path.read_text(encoding="utf-8"))
|
||||
assert payload["host"] == "127.0.0.1"
|
||||
assert payload["cors_origins"] == ""
|
||||
assert payload["log_request_body"] is False
|
||||
assert payload["model_spec_override"].endswith("utils\\audio_cpp\\model_specs") or payload[
|
||||
"model_spec_override"
|
||||
].endswith("utils/audio_cpp/model_specs")
|
||||
assert payload["models"][0]["task"] == "tts"
|
||||
assert Path(payload["models"][0]["path"]).is_absolute()
|
||||
assert runtime.client.voices("managed-pocket") == ["managed"]
|
||||
|
||||
runtime.close()
|
||||
assert process.poll() is not None
|
||||
assert not config_path.exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_external_session_is_keyed_and_never_terminates_server(fake_audio_cpp_server):
|
||||
server, url = fake_audio_cpp_server
|
||||
config = {
|
||||
"connection_mode": "existing_server",
|
||||
"server_url": url,
|
||||
"model_id": "pocket",
|
||||
}
|
||||
first = get_audio_cpp_session(config)
|
||||
second = get_audio_cpp_session(dict(config))
|
||||
assert first is second
|
||||
assert first.owned is False
|
||||
assert first.model_id == "pocket"
|
||||
assert first.model_metadata["family"] == "pocket_tts"
|
||||
assert first.task == "tts"
|
||||
assert first.voices() == ["alba", "cosette"]
|
||||
assert server.model_queries == 1
|
||||
|
||||
first.close()
|
||||
with urllib.request.urlopen(f"{url}/health", timeout=1) as response:
|
||||
assert json.load(response)["status"] == "ok"
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_owned_session_restarts_and_reregisters_after_exact_child_exit(tmp_path, monkeypatch):
|
||||
script = _write_fake_server_script(tmp_path / "fake_audio_cpp_server.py")
|
||||
|
||||
fake_management = types.ModuleType("comfy.model_management")
|
||||
fake_management.current_loaded_models = []
|
||||
fake_management.cleanup_models = lambda: None
|
||||
|
||||
class LoadedModel:
|
||||
def __init__(self, model):
|
||||
self.model = model
|
||||
|
||||
fake_management.LoadedModel = LoadedModel
|
||||
monkeypatch.setitem(sys.modules, "comfy.model_management", fake_management)
|
||||
comfy_module = sys.modules.get("comfy")
|
||||
if comfy_module is not None:
|
||||
monkeypatch.setattr(comfy_module, "model_management", fake_management, raising=False)
|
||||
|
||||
session = get_audio_cpp_session(_owned_config(tmp_path, script))
|
||||
assert session.proxy.model_size() >= len(b"audio-cpp-test-model")
|
||||
first = session.run({"text": "first"})
|
||||
first_runtime = session.process
|
||||
first_process = first_runtime.process
|
||||
assert first.sample_rate == 22050
|
||||
assert len(fake_management.current_loaded_models) == 1
|
||||
|
||||
first_process.kill()
|
||||
first_process.wait(timeout=2)
|
||||
second = session.run({"text": "second"})
|
||||
second_runtime = session.process
|
||||
second_process = second_runtime.process
|
||||
assert second.sample_rate == 22050
|
||||
assert second_runtime is not first_runtime
|
||||
assert second_process is not first_process
|
||||
assert len(fake_management.current_loaded_models) == 1
|
||||
|
||||
session.restart_owned_runtime()
|
||||
reset_process = session.process.process
|
||||
assert second_process.poll() is not None
|
||||
assert reset_process is not second_process
|
||||
assert session.run({"text": "after explicit reset"}).sample_rate == 22050
|
||||
assert len(fake_management.current_loaded_models) == 1
|
||||
|
||||
assert session.proxy.partially_unload("cpu", 1) == 0
|
||||
tracked_model = fake_management.current_loaded_models[0]
|
||||
session.proxy.unpatch_model("cpu")
|
||||
assert reset_process.poll() is not None
|
||||
assert fake_management.current_loaded_models == [tracked_model]
|
||||
fake_management.current_loaded_models.pop(0)
|
||||
|
||||
session.run({"text": "third"})
|
||||
third_process = session.process.process
|
||||
assert third_process.poll() is None
|
||||
assert len(fake_management.current_loaded_models) == 1
|
||||
session.close()
|
||||
assert third_process.poll() is not None
|
||||
assert len(fake_management.current_loaded_models) == 0
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_session_makes_voice_reference_path_absolute(fake_audio_cpp_server, tmp_path, monkeypatch):
|
||||
server, url = fake_audio_cpp_server
|
||||
monkeypatch.chdir(tmp_path)
|
||||
reference = tmp_path / "voice.wav"
|
||||
reference.write_bytes(_wav_bytes())
|
||||
session = get_audio_cpp_session({
|
||||
"connection_mode": "existing_server",
|
||||
"server_url": url,
|
||||
"model_id": "pocket",
|
||||
})
|
||||
|
||||
session.run({"text": "absolute path", "voice_ref": "voice.wav"})
|
||||
|
||||
sent_path = server.requests[-1]["request"]["voice_ref"]
|
||||
assert Path(sent_path).is_absolute()
|
||||
assert Path(sent_path) == reference
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_session_resolver_discovers_existing_model_root(tmp_path, monkeypatch):
|
||||
script = _write_fake_server_script(tmp_path / "fake_audio_cpp_server.py")
|
||||
external_root = tmp_path / "existing-models"
|
||||
managed_root = tmp_path / "suite-managed-models"
|
||||
package = load_catalog().package("pocket_tts_english_q8_0")
|
||||
installed = package_install_path(package, external_root)
|
||||
for relative_path in package.local_files:
|
||||
target = installed / relative_path
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_bytes(b"installed")
|
||||
|
||||
monkeypatch.setattr(
|
||||
audio_cpp_resolver,
|
||||
"load_settings",
|
||||
lambda: AudioCppSettings(
|
||||
connection_mode="managed",
|
||||
model_roots=(str(external_root),),
|
||||
managed_model_root=str(managed_root),
|
||||
runtime_backend="cpu",
|
||||
),
|
||||
)
|
||||
session = get_audio_cpp_session({
|
||||
"connection_mode": "managed",
|
||||
"family": "pocket_tts",
|
||||
"package_id": package.id,
|
||||
"task": "tts",
|
||||
"backend": "cpu",
|
||||
"binary_path": sys.executable,
|
||||
"binary_args": [str(script)],
|
||||
"startup_timeout": 5.0,
|
||||
"connect_timeout": 1.0,
|
||||
"request_timeout": 5.0,
|
||||
})
|
||||
|
||||
result = session.run({"text": "resolved"})
|
||||
|
||||
assert result.sample_rate == 22050
|
||||
assert session.config["model_path"] == str(installed.resolve())
|
||||
assert not managed_root.exists()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_external_session_selects_the_servers_sole_model(fake_audio_cpp_server):
|
||||
server, url = fake_audio_cpp_server
|
||||
|
||||
session = get_audio_cpp_session({
|
||||
"connection_mode": "external_server",
|
||||
"server_url": url,
|
||||
"model_id": "",
|
||||
})
|
||||
reused = get_audio_cpp_session({
|
||||
"connection_mode": "external_server",
|
||||
"server_url": url,
|
||||
"model_id": "",
|
||||
})
|
||||
|
||||
assert reused is session
|
||||
assert session.model_id == "pocket"
|
||||
assert session.model_metadata["family"] == "pocket_tts"
|
||||
assert session.task == "tts"
|
||||
assert server.model_queries == 1
|
||||
|
||||
before = next(item for item in audio_cpp_session_statuses() if item["model_id"] == "pocket")
|
||||
assert before["state"] == "server_ready"
|
||||
assert before["owned"] is False
|
||||
assert before["endpoint"] == url
|
||||
|
||||
session.run({"text": "status"})
|
||||
after = next(item for item in audio_cpp_session_statuses() if item["model_id"] == "pocket")
|
||||
assert after["state"] == "model_ready"
|
||||
@@ -11,6 +11,7 @@ _ADAPTER_MAP: Dict[str, str] = {
|
||||
"qwen3_tts": "engines.adapters.asr_qwen3_adapter.Qwen3ASREngineAdapter",
|
||||
"qwen3": "engines.adapters.asr_qwen3_adapter.Qwen3ASREngineAdapter",
|
||||
"granite_asr": "engines.adapters.asr_granite_adapter.GraniteASREngineAdapter",
|
||||
"audio_cpp": "engines.adapters.asr_audio_cpp_adapter.AudioCppASREngineAdapter",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -702,6 +702,43 @@ class OmniVoiceCacheKeyGenerator(CacheKeyGenerator):
|
||||
return hashlib.md5(cache_string.encode()).hexdigest()
|
||||
|
||||
|
||||
class AudioCppCacheKeyGenerator(CacheKeyGenerator):
|
||||
"""Cache key generator for the generic audio.cpp server backend."""
|
||||
|
||||
def generate_cache_key(self, **params) -> str:
|
||||
cache_data = {
|
||||
'text': params.get('text', ''),
|
||||
'audio_component': params.get('audio_component', ''),
|
||||
'reference_text': params.get('reference_text', ''),
|
||||
'family': params.get('family', ''),
|
||||
'package_id': params.get('package_id', ''),
|
||||
'model_path': params.get('model_path', ''),
|
||||
'model_id': params.get('model_id', ''),
|
||||
'task': params.get('task', ''),
|
||||
'connection_mode': params.get('connection_mode', ''),
|
||||
'server_url': params.get('server_url', ''),
|
||||
'binary_path': params.get('binary_path', ''),
|
||||
'backend': params.get('backend', ''),
|
||||
'device_index': params.get('device_index', 0),
|
||||
'language': params.get('language', ''),
|
||||
'voice_id': params.get('voice_id', ''),
|
||||
'instruct': params.get('instruct', ''),
|
||||
'temperature': params.get('temperature'),
|
||||
'top_p': params.get('top_p'),
|
||||
'top_k': params.get('top_k'),
|
||||
'repetition_penalty': params.get('repetition_penalty'),
|
||||
'max_tokens': params.get('max_tokens'),
|
||||
'max_steps': params.get('max_steps'),
|
||||
'num_inference_steps': params.get('num_inference_steps'),
|
||||
'guidance_scale': params.get('guidance_scale'),
|
||||
'seed': params.get('seed', 0),
|
||||
'request_options': params.get('request_options', ''),
|
||||
'character': params.get('character', 'narrator'),
|
||||
'engine': 'audio_cpp',
|
||||
}
|
||||
return hashlib.md5(str(sorted(cache_data.items())).encode()).hexdigest()
|
||||
|
||||
|
||||
class AudioCache:
|
||||
"""Unified audio cache manager for all TTS engines."""
|
||||
|
||||
@@ -720,6 +757,7 @@ class AudioCache:
|
||||
'dots_tts': DotsTTSCacheKeyGenerator(),
|
||||
'dramabox': DramaBoxCacheKeyGenerator(),
|
||||
'fish_audio_s2': FishAudioS2CacheKeyGenerator(),
|
||||
'audio_cpp': AudioCppCacheKeyGenerator(),
|
||||
'omnivoice': OmniVoiceCacheKeyGenerator(),
|
||||
'moss_tts': MossTTSCacheKeyGenerator(),
|
||||
'moss_soundeffect_v2': MossSoundEffectV2CacheKeyGenerator(),
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
"""Pinned audio.cpp integration data, discovery, and installation helpers."""
|
||||
|
||||
from .catalog import (
|
||||
AUDIO_CPP_RELEASE_COMMIT,
|
||||
AUDIO_CPP_RELEASE_TAG,
|
||||
AUDIO_CPP_RELEASE_VERSION,
|
||||
AudioCppCatalog,
|
||||
FamilyRecord,
|
||||
PackageRecord,
|
||||
family_choices,
|
||||
get_family,
|
||||
get_model_specs_dir,
|
||||
get_package,
|
||||
load_catalog,
|
||||
package_choices,
|
||||
recommended_package,
|
||||
resolve_task,
|
||||
)
|
||||
from .settings import AudioCppSettings, get_settings_path, load_settings, save_settings
|
||||
from .resolver import AudioCppResolutionError, resolve_audio_cpp_config
|
||||
|
||||
__all__ = [
|
||||
"AUDIO_CPP_RELEASE_COMMIT",
|
||||
"AUDIO_CPP_RELEASE_TAG",
|
||||
"AUDIO_CPP_RELEASE_VERSION",
|
||||
"AudioCppCatalog",
|
||||
"AudioCppSettings",
|
||||
"AudioCppResolutionError",
|
||||
"FamilyRecord",
|
||||
"PackageRecord",
|
||||
"family_choices",
|
||||
"get_family",
|
||||
"get_model_specs_dir",
|
||||
"get_package",
|
||||
"get_settings_path",
|
||||
"load_catalog",
|
||||
"load_settings",
|
||||
"package_choices",
|
||||
"recommended_package",
|
||||
"resolve_audio_cpp_config",
|
||||
"resolve_task",
|
||||
"save_settings",
|
||||
]
|
||||
@@ -0,0 +1,184 @@
|
||||
"""Resolve Suite integration capabilities for pinned audio.cpp families."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Mapping
|
||||
|
||||
import yaml
|
||||
|
||||
from .catalog import AUDIO_CPP_RELEASE_TAG, PackageRecord, load_catalog
|
||||
|
||||
|
||||
class CapabilityError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def get_capability_path() -> Path:
|
||||
return Path(__file__).resolve().with_name("integration_capabilities.yaml")
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _load_overlay() -> Dict[str, Any]:
|
||||
path = get_capability_path()
|
||||
try:
|
||||
raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
||||
except (OSError, yaml.YAMLError) as exc:
|
||||
raise CapabilityError(f"Cannot read audio.cpp capability overlay {path}: {exc}") from exc
|
||||
if raw.get("schema_version") != 1 or raw.get("release") != AUDIO_CPP_RELEASE_TAG:
|
||||
raise CapabilityError("audio.cpp capability overlay release/schema does not match the catalog")
|
||||
return dict(raw)
|
||||
|
||||
|
||||
def _merge(base: Mapping[str, Any], override: Mapping[str, Any]) -> Dict[str, Any]:
|
||||
value = deepcopy(dict(base))
|
||||
for key, item in override.items():
|
||||
if isinstance(item, Mapping) and isinstance(value.get(key), Mapping):
|
||||
value[key] = _merge(value[key], item)
|
||||
else:
|
||||
value[key] = deepcopy(item)
|
||||
return value
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def load_capabilities() -> Dict[str, Dict[str, Any]]:
|
||||
raw = _load_overlay()
|
||||
catalog = load_catalog()
|
||||
families = raw.get("families") or {}
|
||||
if set(families) != set(catalog.families):
|
||||
missing = sorted(set(catalog.families) - set(families))
|
||||
extra = sorted(set(families) - set(catalog.families))
|
||||
raise CapabilityError(f"audio.cpp capability families mismatch; missing={missing}, extra={extra}")
|
||||
|
||||
resolved: Dict[str, Dict[str, Any]] = {}
|
||||
defaults = raw.get("defaults") or {}
|
||||
for family_id, override in families.items():
|
||||
family = catalog.family(family_id)
|
||||
item = _merge(defaults, override or {})
|
||||
item.update(
|
||||
id=family.id,
|
||||
display_name=family.display_name,
|
||||
description=family.description,
|
||||
languages=list(family.languages),
|
||||
upstream_tasks=list(family.runtime_tasks),
|
||||
packages=[package.id for package in family.packages],
|
||||
recommended_package_id=family.recommended_package_id,
|
||||
)
|
||||
asr_features = item.get("asr_features") or {}
|
||||
if not isinstance(asr_features, Mapping):
|
||||
raise CapabilityError(
|
||||
f"audio.cpp {family_id} asr_features must be an object"
|
||||
)
|
||||
diarization = str(asr_features.get("diarization", "none"))
|
||||
timing = str(asr_features.get("timing", "none"))
|
||||
if diarization not in {"none", "native"}:
|
||||
raise CapabilityError(
|
||||
f"audio.cpp {family_id} has invalid ASR diarization capability: {diarization}"
|
||||
)
|
||||
if timing not in {
|
||||
"none",
|
||||
"native_word",
|
||||
"native_segment",
|
||||
"optional_forced_aligner",
|
||||
}:
|
||||
raise CapabilityError(
|
||||
f"audio.cpp {family_id} has invalid ASR timing capability: {timing}"
|
||||
)
|
||||
resolved[family_id] = item
|
||||
return resolved
|
||||
|
||||
|
||||
def get_capability(family: str) -> Dict[str, Any]:
|
||||
try:
|
||||
return deepcopy(load_capabilities()[str(family)])
|
||||
except KeyError as exc:
|
||||
raise CapabilityError(f"Unknown audio.cpp capability family: {family!r}") from exc
|
||||
|
||||
|
||||
def public_capabilities() -> Dict[str, Any]:
|
||||
catalog = load_catalog()
|
||||
raw_sizes = _load_overlay().get("package_sizes") or {}
|
||||
if set(raw_sizes) != set(catalog.packages):
|
||||
missing = sorted(set(catalog.packages) - set(raw_sizes))
|
||||
extra = sorted(set(raw_sizes) - set(catalog.packages))
|
||||
raise CapabilityError(f"audio.cpp package sizes mismatch; missing={missing}, extra={extra}")
|
||||
packages = {}
|
||||
for package_id, package in catalog.packages.items():
|
||||
size = raw_sizes[package_id]
|
||||
if not isinstance(size, int) or size <= 0:
|
||||
raise CapabilityError(f"Invalid estimated size for audio.cpp package {package_id!r}")
|
||||
dependencies = get_package_dependencies(package_id)
|
||||
dependency_bytes = sum(
|
||||
int(item["estimated_download_bytes"]) for item in dependencies
|
||||
)
|
||||
packages[package_id] = {
|
||||
"id": package.id,
|
||||
"family": package.family,
|
||||
"display_name": package.display_name,
|
||||
"format": package.format,
|
||||
"precision": package.precision,
|
||||
"estimated_download_bytes": size + dependency_bytes,
|
||||
"primary_download_bytes": size,
|
||||
"dependencies": [item["package"].id for item in dependencies],
|
||||
}
|
||||
return {
|
||||
"schema_version": 1,
|
||||
"release": AUDIO_CPP_RELEASE_TAG,
|
||||
"sizes_checked_at": "2026-08-13",
|
||||
"families": load_capabilities(),
|
||||
"packages": packages,
|
||||
}
|
||||
|
||||
|
||||
def get_package_dependencies(package_id: str) -> list[Dict[str, Any]]:
|
||||
entries = (_load_overlay().get("package_dependencies") or {}).get(package_id, [])
|
||||
dependencies = []
|
||||
for entry in entries:
|
||||
package = PackageRecord(
|
||||
family=str(entry["family"]),
|
||||
id=str(entry["id"]),
|
||||
display_name=str(entry["display_name"]),
|
||||
target_directory=str(entry["target_directory"]),
|
||||
format=str(entry["format"]),
|
||||
precision=str(entry["precision"]),
|
||||
files=tuple(str(value) for value in entry["files"]),
|
||||
strip_prefix=str(entry.get("strip_prefix", "")),
|
||||
download={
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": str(entry["repo"]),
|
||||
"revision": str(entry.get("revision", "main")),
|
||||
"gated": False,
|
||||
},
|
||||
)
|
||||
# Validate local mappings before the downloader touches disk.
|
||||
package.local_files
|
||||
dependencies.append(
|
||||
{
|
||||
"package": package,
|
||||
"session_option": str(entry["session_option"]),
|
||||
"estimated_download_bytes": int(entry["estimated_download_bytes"]),
|
||||
}
|
||||
)
|
||||
return dependencies
|
||||
|
||||
|
||||
def validate_voice_reference(family: str, voice_ref: Any, character: str = "narrator") -> None:
|
||||
"""Enforce requirements that the frontend panel merely explains."""
|
||||
from utils.voice.reference import effective_voice_audio
|
||||
|
||||
capability = get_capability(family)
|
||||
audio_requirement = capability["reference_audio"]
|
||||
has_audio = isinstance(voice_ref, Mapping) and effective_voice_audio(voice_ref) is not None
|
||||
if audio_requirement in {"required", "required_per_speaker"} and not has_audio:
|
||||
raise ValueError(f"audio.cpp {family} requires reference audio for '{character}'")
|
||||
transcript = ""
|
||||
if isinstance(voice_ref, Mapping):
|
||||
transcript = str(
|
||||
voice_ref.get("reference_text") or voice_ref.get("prompt_text") or voice_ref.get("text") or ""
|
||||
).strip()
|
||||
if capability["reference_transcript"] == "required" and not transcript:
|
||||
raise ValueError(
|
||||
f"audio.cpp {family} requires the transcript matching '{character}' reference audio"
|
||||
)
|
||||
@@ -0,0 +1,38 @@
|
||||
"""HTTP route for the audio.cpp engine capability panel."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def register_audio_cpp_capability_routes(routes, web) -> None:
|
||||
@routes.get("/api/tts-audio-suite/audio-cpp-capabilities")
|
||||
async def get_audio_cpp_capabilities(_request):
|
||||
try:
|
||||
from .capabilities import public_capabilities
|
||||
|
||||
return web.json_response(public_capabilities())
|
||||
except Exception as exc:
|
||||
return web.json_response({"error": str(exc)}, status=500)
|
||||
|
||||
@routes.get("/api/tts-audio-suite/audio-cpp-status")
|
||||
async def get_audio_cpp_status(_request):
|
||||
try:
|
||||
from .session import audio_cpp_session_statuses
|
||||
|
||||
return web.json_response({"sessions": audio_cpp_session_statuses()})
|
||||
except Exception as exc:
|
||||
return web.json_response({"sessions": [], "error": str(exc)}, status=500)
|
||||
|
||||
@routes.post("/api/tts-audio-suite/audio-cpp-stop")
|
||||
async def stop_audio_cpp_session(request):
|
||||
try:
|
||||
from .session import stop_owned_audio_cpp_session
|
||||
|
||||
data = await request.json()
|
||||
stopped = stop_owned_audio_cpp_session(str(data.get("session_id", "")))
|
||||
if not stopped:
|
||||
return web.json_response({"error": "audio.cpp session not found"}, status=404)
|
||||
return web.json_response({"status": "stopped"})
|
||||
except PermissionError as exc:
|
||||
return web.json_response({"error": str(exc)}, status=403)
|
||||
except Exception as exc:
|
||||
return web.json_response({"error": str(exc)}, status=500)
|
||||
@@ -0,0 +1,368 @@
|
||||
"""Pinned audio.cpp release-0.5.1 Suite-compatible model catalog.
|
||||
|
||||
The bundled JSON files are exact copies of the selected upstream tag's model
|
||||
specifications. The executable's compiled task IDs are kept separately because
|
||||
several release specs expose broader or differently-spelled task metadata.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from functools import lru_cache
|
||||
from pathlib import Path, PurePosixPath
|
||||
from typing import Any, Dict, Iterator, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
|
||||
AUDIO_CPP_RELEASE_VERSION = "0.5.1"
|
||||
AUDIO_CPP_RELEASE_TAG = "release-0.5.1"
|
||||
AUDIO_CPP_RELEASE_COMMIT = "238ab6a9e321c17de8e120559f57efeedaeb1345"
|
||||
|
||||
MODEL_SPEC_FILENAMES: Tuple[str, ...] = (
|
||||
"citrinet_asr.json",
|
||||
"chatterbox.json",
|
||||
"confucius4_tts.json",
|
||||
"dramabox.json",
|
||||
"fish_audio.json",
|
||||
"fun_asr_nano.json",
|
||||
"glm_tts.json",
|
||||
"higgs_audio_tts.json",
|
||||
"higgs_audio_stt.json",
|
||||
"hviske_asr.json",
|
||||
"index_tts2.json",
|
||||
"inflect_v2.json",
|
||||
"irodori_tts.json",
|
||||
"kroko_asr.json",
|
||||
"miotts.json",
|
||||
"moss_tts_local.json",
|
||||
"moss_tts_nano.json",
|
||||
"nemotron_asr.json",
|
||||
"omnivoice.json",
|
||||
"outetts.json",
|
||||
"parakeet_tdt.json",
|
||||
"pocket_tts.json",
|
||||
"qwen3_tts.json",
|
||||
"qwen3_asr.json",
|
||||
"seed_vc.json",
|
||||
"supertonic.json",
|
||||
"vevo2.json",
|
||||
"vibevoice.json",
|
||||
"vibevoice_asr.json",
|
||||
"vietneu_tts.json",
|
||||
"voxcpm2.json",
|
||||
"voxtral_realtime.json",
|
||||
)
|
||||
|
||||
# These are the task IDs actually compiled into release-0.5.1. Do not derive
|
||||
# them from the broader human-facing ``tasks`` arrays in the JSON specs.
|
||||
COMPILED_TASKS: Mapping[str, Tuple[str, ...]] = {
|
||||
"citrinet_asr": ("asr",),
|
||||
"chatterbox": ("clon", "vc"),
|
||||
"confucius4_tts": ("clon",),
|
||||
"dramabox": ("tts", "clon"),
|
||||
"fish_audio": ("tts",),
|
||||
"fun_asr_nano": ("asr",),
|
||||
"glm_tts": ("tts", "clon"),
|
||||
"higgs_audio_tts": ("tts",),
|
||||
"higgs_audio_stt": ("asr",),
|
||||
"hviske_asr": ("asr",),
|
||||
"index_tts2": ("tts", "clon"),
|
||||
"inflect_v2": ("tts",),
|
||||
"irodori_tts": ("tts", "clon", "vdes"),
|
||||
"kroko_asr": ("asr",),
|
||||
"miotts": ("tts",),
|
||||
"moss_tts_local": ("tts", "clon"),
|
||||
"moss_tts_nano": ("tts", "clon"),
|
||||
"nemotron_asr": ("asr",),
|
||||
"omnivoice": ("tts",),
|
||||
"outetts": ("tts", "clon"),
|
||||
"parakeet_tdt": ("asr",),
|
||||
"pocket_tts": ("tts",),
|
||||
"qwen3_tts": ("tts", "vdes"),
|
||||
"qwen3_asr": ("asr",),
|
||||
"seed_vc": ("vc", "svc"),
|
||||
"supertonic": ("tts",),
|
||||
"vevo2": ("tts", "vc", "s2s", "svc"),
|
||||
"vibevoice": ("tts",),
|
||||
"vibevoice_asr": ("asr",),
|
||||
"vietneu_tts": ("tts", "vdes"),
|
||||
"voxcpm2": ("tts",),
|
||||
"voxtral_realtime": ("asr",),
|
||||
}
|
||||
|
||||
_TASK_ALIASES = {
|
||||
"clone": "clon",
|
||||
"cloning": "clon",
|
||||
"voice_clone": "clon",
|
||||
"voice_cloning": "clon",
|
||||
"voice_design": "vdes",
|
||||
"design": "vdes",
|
||||
"voice_conversion": "vc",
|
||||
"speech_to_speech": "s2s",
|
||||
"singing_voice_conversion": "svc",
|
||||
}
|
||||
|
||||
|
||||
class CatalogError(ValueError):
|
||||
"""Raised when a pinned model specification is missing or inconsistent."""
|
||||
|
||||
|
||||
def _safe_package_relative_path(value: str, label: str) -> PurePosixPath:
|
||||
normalized = value.replace("\\", "/")
|
||||
path = PurePosixPath(normalized)
|
||||
if (
|
||||
not normalized
|
||||
or path.is_absolute()
|
||||
or ".." in path.parts
|
||||
or (path.parts and ":" in path.parts[0])
|
||||
):
|
||||
raise CatalogError(f"Unsafe {label}: {value!r}")
|
||||
return path
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PackageRecord:
|
||||
family: str
|
||||
id: str
|
||||
display_name: str
|
||||
target_directory: str
|
||||
format: str
|
||||
precision: str
|
||||
files: Tuple[str, ...]
|
||||
strip_prefix: str
|
||||
download: Mapping[str, Any]
|
||||
default: bool = False
|
||||
|
||||
def local_relative_path(self, remote_path: str) -> Path:
|
||||
"""Map one remote package path to its installed relative path."""
|
||||
|
||||
remote = _safe_package_relative_path(remote_path, "remote file path")
|
||||
prefix_text = self.strip_prefix.replace("\\", "/").rstrip("/")
|
||||
if prefix_text in ("", "."):
|
||||
local = remote
|
||||
else:
|
||||
prefix = _safe_package_relative_path(prefix_text, "strip_prefix")
|
||||
if remote == prefix or remote.parts[: len(prefix.parts)] != prefix.parts:
|
||||
raise CatalogError(
|
||||
f"Package {self.id!r} file {remote_path!r} is outside "
|
||||
f"strip_prefix {self.strip_prefix!r}"
|
||||
)
|
||||
local = PurePosixPath(*remote.parts[len(prefix.parts) :])
|
||||
if not local.parts:
|
||||
raise CatalogError(f"Package {self.id!r} maps {remote_path!r} to an empty path")
|
||||
return Path(*local.parts)
|
||||
|
||||
@property
|
||||
def local_files(self) -> Tuple[Path, ...]:
|
||||
return tuple(self.local_relative_path(path) for path in self.files)
|
||||
|
||||
@property
|
||||
def repo(self) -> str:
|
||||
return str(self.download.get("repo", ""))
|
||||
|
||||
@property
|
||||
def revision(self) -> str:
|
||||
return str(self.download.get("revision", "main"))
|
||||
|
||||
@property
|
||||
def gated(self) -> bool:
|
||||
return bool(self.download.get("gated", False))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FamilyRecord:
|
||||
id: str
|
||||
display_name: str
|
||||
description: str
|
||||
category: str
|
||||
status: str
|
||||
runtime_tasks: Tuple[str, ...]
|
||||
languages: Tuple[str, ...]
|
||||
options: Mapping[str, Any]
|
||||
packages: Tuple[PackageRecord, ...]
|
||||
recommended_package_id: str
|
||||
spec_filename: str
|
||||
raw: Mapping[str, Any]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AudioCppCatalog:
|
||||
families: Mapping[str, FamilyRecord]
|
||||
packages: Mapping[str, PackageRecord]
|
||||
specs_dir: Path
|
||||
release_version: str = AUDIO_CPP_RELEASE_VERSION
|
||||
|
||||
def iter_families(self) -> Iterator[FamilyRecord]:
|
||||
return iter(self.families.values())
|
||||
|
||||
def iter_packages(self, family: Optional[str] = None) -> Iterator[PackageRecord]:
|
||||
if family is None:
|
||||
return iter(self.packages.values())
|
||||
return iter(self.family(family).packages)
|
||||
|
||||
def family(self, family_id: str) -> FamilyRecord:
|
||||
try:
|
||||
return self.families[family_id]
|
||||
except KeyError as exc:
|
||||
raise CatalogError(f"Unknown audio.cpp family: {family_id!r}") from exc
|
||||
|
||||
def package(self, package_id: str) -> PackageRecord:
|
||||
try:
|
||||
return self.packages[package_id]
|
||||
except KeyError as exc:
|
||||
raise CatalogError(f"Unknown audio.cpp package: {package_id!r}") from exc
|
||||
|
||||
|
||||
def get_model_specs_dir() -> Path:
|
||||
"""Return the exact release-0.5.1 spec directory shipped with the node."""
|
||||
|
||||
return Path(__file__).resolve().parent / "model_specs"
|
||||
|
||||
|
||||
def _merged_download(spec: Mapping[str, Any], package: Mapping[str, Any]) -> Dict[str, Any]:
|
||||
merged = dict(spec.get("package_defaults", {}).get("download", {}))
|
||||
merged.update(package.get("download", {}))
|
||||
return merged
|
||||
|
||||
|
||||
def _load_catalog(specs_dir: Path) -> AudioCppCatalog:
|
||||
families: Dict[str, FamilyRecord] = {}
|
||||
packages: Dict[str, PackageRecord] = {}
|
||||
|
||||
for filename in MODEL_SPEC_FILENAMES:
|
||||
path = specs_dir / filename
|
||||
if not path.is_file():
|
||||
raise CatalogError(f"Missing pinned audio.cpp model spec: {path}")
|
||||
try:
|
||||
raw = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise CatalogError(f"Cannot read audio.cpp model spec {path}: {exc}") from exc
|
||||
|
||||
family_id = str(raw.get("family", ""))
|
||||
if family_id not in COMPILED_TASKS:
|
||||
raise CatalogError(f"Spec {filename} has unsupported release family {family_id!r}")
|
||||
if family_id in families:
|
||||
raise CatalogError(f"Duplicate audio.cpp family {family_id!r}")
|
||||
|
||||
family_packages = []
|
||||
for package_data in raw.get("packages", []):
|
||||
package_id = str(package_data.get("id", ""))
|
||||
if not package_id or package_id in packages:
|
||||
raise CatalogError(f"Missing or duplicate package ID {package_id!r} in {filename}")
|
||||
package = PackageRecord(
|
||||
family=family_id,
|
||||
id=package_id,
|
||||
display_name=str(package_data.get("display_name", package_id)),
|
||||
target_directory=str(package_data.get("target_directory", "")),
|
||||
format=str(package_data.get("format", "")),
|
||||
precision=str(package_data.get("precision", "")),
|
||||
files=tuple(str(item) for item in package_data.get("files", [])),
|
||||
strip_prefix=str(package_data.get("strip_prefix", "")),
|
||||
download=_merged_download(raw, package_data),
|
||||
default=bool(package_data.get("default", False)),
|
||||
)
|
||||
_safe_package_relative_path(package.target_directory, "target_directory")
|
||||
if not package.files:
|
||||
raise CatalogError(f"Package {package_id!r} has no downloadable files")
|
||||
# Validate prefix mappings when loading, before any filesystem mutation.
|
||||
package.local_files
|
||||
if package.download.get("kind") != "huggingface_snapshot" or not package.repo:
|
||||
raise CatalogError(f"Package {package_id!r} has no supported download source")
|
||||
packages[package_id] = package
|
||||
family_packages.append(package)
|
||||
|
||||
recommendation = str(raw.get("ui", {}).get("recommended_package", ""))
|
||||
if not recommendation:
|
||||
recommendation = next((p.id for p in family_packages if p.default), "")
|
||||
if recommendation not in {package.id for package in family_packages}:
|
||||
raise CatalogError(
|
||||
f"Family {family_id!r} recommends unknown package {recommendation!r}"
|
||||
)
|
||||
|
||||
families[family_id] = FamilyRecord(
|
||||
id=family_id,
|
||||
display_name=str(raw.get("display_name", family_id)),
|
||||
description=str(raw.get("description", "")),
|
||||
category=str(raw.get("category", "")),
|
||||
status=str(raw.get("status", "")),
|
||||
runtime_tasks=COMPILED_TASKS[family_id],
|
||||
languages=tuple(str(item) for item in raw.get("languages", [])),
|
||||
options=dict(raw.get("options", {})),
|
||||
packages=tuple(family_packages),
|
||||
recommended_package_id=recommendation,
|
||||
spec_filename=filename,
|
||||
raw=raw,
|
||||
)
|
||||
|
||||
if set(families) != set(COMPILED_TASKS):
|
||||
missing = sorted(set(COMPILED_TASKS) - set(families))
|
||||
raise CatalogError(f"Pinned audio.cpp catalog is incomplete; missing {missing}")
|
||||
if len(packages) != 96:
|
||||
raise CatalogError(f"Expected 96 Suite-compatible release-0.5.1 packages, found {len(packages)}")
|
||||
return AudioCppCatalog(families=families, packages=packages, specs_dir=specs_dir)
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _load_bundled_catalog() -> AudioCppCatalog:
|
||||
return _load_catalog(get_model_specs_dir())
|
||||
|
||||
|
||||
def load_catalog(specs_dir: Optional[Path] = None) -> AudioCppCatalog:
|
||||
"""Load the bundled catalog, or validate an equivalent override directory."""
|
||||
|
||||
if specs_dir is None:
|
||||
return _load_bundled_catalog()
|
||||
return _load_catalog(Path(specs_dir).expanduser().resolve())
|
||||
|
||||
|
||||
def family_choices() -> list[str]:
|
||||
return list(load_catalog().families)
|
||||
|
||||
|
||||
def package_choices(family: Optional[str] = None) -> list[str]:
|
||||
catalog = load_catalog()
|
||||
if family is None:
|
||||
return list(catalog.packages)
|
||||
return [package.id for package in catalog.family(family).packages]
|
||||
|
||||
|
||||
def get_family(family: str) -> FamilyRecord:
|
||||
return load_catalog().family(family)
|
||||
|
||||
|
||||
def get_package(package_id: str) -> PackageRecord:
|
||||
return load_catalog().package(package_id)
|
||||
|
||||
|
||||
def recommended_package(family: str) -> str:
|
||||
return get_family(family).recommended_package_id
|
||||
|
||||
|
||||
def resolve_task(family: str, package_id: Optional[str], requested: str = "auto") -> str:
|
||||
"""Resolve a UI task name to a release-0.5.1 compiled task ID."""
|
||||
|
||||
catalog = load_catalog()
|
||||
family_record = catalog.family(family)
|
||||
package = catalog.package(package_id) if package_id is not None else None
|
||||
if package is not None and package.family != family:
|
||||
raise CatalogError(f"Package {package_id!r} does not belong to family {family!r}")
|
||||
normalized = str(requested or "auto").strip().lower().replace("-", "_").replace(" ", "_")
|
||||
if normalized == "auto":
|
||||
if package is not None and "vdes" in family_record.runtime_tasks:
|
||||
package_label = f"{package.id} {package.display_name}".lower().replace("_", "")
|
||||
if "voicedesign" in package_label:
|
||||
return "vdes"
|
||||
return family_record.runtime_tasks[0]
|
||||
normalized = _TASK_ALIASES.get(normalized, normalized)
|
||||
# Several release families condition cloning through a speaker reference on
|
||||
# the compiled ``tts`` task instead of exposing a separate ``clon`` task.
|
||||
if normalized == "clon" and "clon" not in family_record.runtime_tasks:
|
||||
if "tts" in family_record.runtime_tasks:
|
||||
return "tts"
|
||||
if normalized not in family_record.runtime_tasks:
|
||||
supported = ", ".join(family_record.runtime_tasks)
|
||||
raise CatalogError(
|
||||
f"Task {requested!r} is unavailable for {family!r} in audio.cpp "
|
||||
f"{AUDIO_CPP_RELEASE_VERSION}; supported: {supported}"
|
||||
)
|
||||
return normalized
|
||||
@@ -0,0 +1,431 @@
|
||||
"""Small stdlib HTTP client for the audio.cpp server API."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import array
|
||||
import base64
|
||||
import binascii
|
||||
import io
|
||||
import json
|
||||
import socket
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
import wave
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, Mapping, Optional
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
class AudioCppClientError(RuntimeError):
|
||||
"""Base error for audio.cpp transport failures."""
|
||||
|
||||
|
||||
class AudioCppConnectionError(AudioCppClientError):
|
||||
"""The audio.cpp endpoint could not be reached."""
|
||||
|
||||
|
||||
class AudioCppTimeoutError(AudioCppClientError):
|
||||
"""The audio.cpp endpoint did not respond before the configured timeout."""
|
||||
|
||||
|
||||
class AudioCppProtocolError(AudioCppClientError):
|
||||
"""The server returned a response that does not match its API contract."""
|
||||
|
||||
|
||||
class AudioCppHTTPError(AudioCppClientError):
|
||||
"""Structured non-success response from audio.cpp."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
status: int,
|
||||
message: str,
|
||||
*,
|
||||
error_type: Optional[str] = None,
|
||||
path: str = "",
|
||||
response_body: str = "",
|
||||
) -> None:
|
||||
self.status = int(status)
|
||||
self.error_type = error_type
|
||||
self.path = path
|
||||
self.response_body = response_body
|
||||
label = f"audio.cpp HTTP {self.status}"
|
||||
if error_type:
|
||||
label += f" ({error_type})"
|
||||
if path:
|
||||
label += f" for {path}"
|
||||
super().__init__(f"{label}: {message}")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AudioCppAudio:
|
||||
"""Decoded PCM audio returned by audio.cpp."""
|
||||
|
||||
waveform: torch.Tensor
|
||||
sample_rate: int
|
||||
channels: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AudioCppTaskResult:
|
||||
"""Decoded result from ``POST /v1/tasks/run``."""
|
||||
|
||||
waveform: Optional[torch.Tensor] = None
|
||||
sample_rate: Optional[int] = None
|
||||
channels: Optional[int] = None
|
||||
named_audio: Dict[str, AudioCppAudio] = field(default_factory=dict)
|
||||
raw: Dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
def _decode_pcm_wav(wav_bytes: bytes, *, context: str = "audio") -> AudioCppAudio:
|
||||
if not isinstance(wav_bytes, (bytes, bytearray)) or not wav_bytes:
|
||||
raise AudioCppProtocolError(f"audio.cpp returned empty {context} WAV data")
|
||||
|
||||
try:
|
||||
with wave.open(io.BytesIO(bytes(wav_bytes)), "rb") as wav_file:
|
||||
channels = int(wav_file.getnchannels())
|
||||
sample_rate = int(wav_file.getframerate())
|
||||
sample_width = int(wav_file.getsampwidth())
|
||||
frame_count = int(wav_file.getnframes())
|
||||
compression = wav_file.getcomptype()
|
||||
frames = wav_file.readframes(frame_count)
|
||||
except (EOFError, wave.Error) as exc:
|
||||
raise AudioCppProtocolError(f"audio.cpp returned an invalid {context} WAV: {exc}") from exc
|
||||
|
||||
if compression != "NONE":
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp returned unsupported compressed {context} WAV data ({compression})"
|
||||
)
|
||||
if channels <= 0 or sample_rate <= 0:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp returned invalid {context} WAV metadata "
|
||||
f"(sample_rate={sample_rate}, channels={channels})"
|
||||
)
|
||||
if sample_width not in (1, 2, 3, 4):
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp returned unsupported {context} WAV sample width: {sample_width} bytes"
|
||||
)
|
||||
|
||||
expected_samples = frame_count * channels
|
||||
expected_bytes = expected_samples * sample_width
|
||||
if len(frames) != expected_bytes:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp returned truncated {context} WAV data "
|
||||
f"({len(frames)} bytes, expected {expected_bytes})"
|
||||
)
|
||||
|
||||
if sample_width == 1:
|
||||
values = torch.tensor(list(frames), dtype=torch.float32)
|
||||
values = (values - 128.0) / 128.0
|
||||
elif sample_width == 2:
|
||||
pcm = array.array("h")
|
||||
pcm.frombytes(frames)
|
||||
if sys.byteorder != "little":
|
||||
pcm.byteswap()
|
||||
values = torch.tensor(pcm, dtype=torch.float32) / 32768.0
|
||||
elif sample_width == 3:
|
||||
decoded = []
|
||||
for offset in range(0, len(frames), 3):
|
||||
sample = int.from_bytes(frames[offset : offset + 3], "little", signed=False)
|
||||
if sample & 0x800000:
|
||||
sample -= 0x1000000
|
||||
decoded.append(sample)
|
||||
values = torch.tensor(decoded, dtype=torch.float32) / 8388608.0
|
||||
else:
|
||||
pcm = array.array("i")
|
||||
pcm.frombytes(frames)
|
||||
if sys.byteorder != "little":
|
||||
pcm.byteswap()
|
||||
values = torch.tensor(pcm, dtype=torch.float32) / 2147483648.0
|
||||
|
||||
if expected_samples == 0:
|
||||
waveform = torch.empty((channels, 0), dtype=torch.float32)
|
||||
else:
|
||||
waveform = values.reshape(frame_count, channels).transpose(0, 1).contiguous()
|
||||
return AudioCppAudio(
|
||||
waveform=waveform.cpu(),
|
||||
sample_rate=sample_rate,
|
||||
channels=channels,
|
||||
)
|
||||
|
||||
|
||||
def _decode_base64_wav(value: Any, *, context: str) -> AudioCppAudio:
|
||||
if not isinstance(value, str) or not value:
|
||||
raise AudioCppProtocolError(f"audio.cpp response field '{context}' must be base64 WAV text")
|
||||
try:
|
||||
wav_bytes = base64.b64decode(value, validate=True)
|
||||
except (ValueError, binascii.Error) as exc:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp response field '{context}' is not valid base64"
|
||||
) from exc
|
||||
return _decode_pcm_wav(wav_bytes, context=context)
|
||||
|
||||
|
||||
class AudioCppClient:
|
||||
"""Synchronous audio.cpp HTTP client using only Python's standard library."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
base_url: str,
|
||||
*,
|
||||
connect_timeout: float = 5.0,
|
||||
request_timeout: float = 600.0,
|
||||
max_response_bytes: int = 2 * 1024 * 1024 * 1024,
|
||||
) -> None:
|
||||
parsed = urllib.parse.urlsplit(str(base_url).strip())
|
||||
if parsed.scheme not in ("http", "https") or not parsed.netloc:
|
||||
raise ValueError(f"Invalid audio.cpp server URL: {base_url!r}")
|
||||
if parsed.query or parsed.fragment:
|
||||
raise ValueError("audio.cpp server URL must not contain a query or fragment")
|
||||
self.base_url = urllib.parse.urlunsplit(
|
||||
(parsed.scheme, parsed.netloc, parsed.path.rstrip("/"), "", "")
|
||||
)
|
||||
self.connect_timeout = max(0.01, float(connect_timeout))
|
||||
self.request_timeout = max(0.01, float(request_timeout))
|
||||
self.max_response_bytes = max(1, int(max_response_bytes))
|
||||
|
||||
def _url(self, path: str) -> str:
|
||||
if not path.startswith("/"):
|
||||
path = "/" + path
|
||||
return self.base_url + path
|
||||
|
||||
@staticmethod
|
||||
def _error_details(body: bytes, fallback: str) -> tuple[str, Optional[str], str]:
|
||||
text = body.decode("utf-8", errors="replace")
|
||||
message = fallback
|
||||
error_type = None
|
||||
try:
|
||||
payload = json.loads(text)
|
||||
error = payload.get("error") if isinstance(payload, dict) else None
|
||||
if isinstance(error, dict):
|
||||
message = str(error.get("message") or fallback)
|
||||
error_type = str(error.get("type")) if error.get("type") else None
|
||||
elif error:
|
||||
message = str(error)
|
||||
except (TypeError, ValueError):
|
||||
if text.strip():
|
||||
message = text.strip()[:1000]
|
||||
return message, error_type, text
|
||||
|
||||
@staticmethod
|
||||
def _is_timeout_error(exc: BaseException) -> bool:
|
||||
if isinstance(exc, (TimeoutError, socket.timeout)):
|
||||
return True
|
||||
if isinstance(exc, urllib.error.URLError):
|
||||
return isinstance(exc.reason, (TimeoutError, socket.timeout))
|
||||
return False
|
||||
|
||||
def _request_bytes(
|
||||
self,
|
||||
method: str,
|
||||
path: str,
|
||||
*,
|
||||
payload: Optional[Mapping[str, Any]] = None,
|
||||
timeout: Optional[float] = None,
|
||||
) -> tuple[bytes, Mapping[str, str]]:
|
||||
body = None
|
||||
headers = {"Accept": "application/json"}
|
||||
if payload is not None:
|
||||
try:
|
||||
body = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ValueError(f"audio.cpp request is not JSON serializable: {exc}") from exc
|
||||
headers["Content-Type"] = "application/json"
|
||||
|
||||
request = urllib.request.Request(
|
||||
self._url(path),
|
||||
data=body,
|
||||
headers=headers,
|
||||
method=method.upper(),
|
||||
)
|
||||
effective_timeout = self.request_timeout if timeout is None else max(0.01, float(timeout))
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=effective_timeout) as response:
|
||||
content_length = response.headers.get("Content-Length")
|
||||
if content_length:
|
||||
try:
|
||||
if int(content_length) > self.max_response_bytes:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp response exceeds {self.max_response_bytes} bytes"
|
||||
)
|
||||
except ValueError:
|
||||
pass
|
||||
response_body = response.read(self.max_response_bytes + 1)
|
||||
if len(response_body) > self.max_response_bytes:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp response exceeds {self.max_response_bytes} bytes"
|
||||
)
|
||||
return response_body, response.headers
|
||||
except urllib.error.HTTPError as exc:
|
||||
error_body = exc.read(65536)
|
||||
message, error_type, response_text = self._error_details(error_body, str(exc.reason))
|
||||
raise AudioCppHTTPError(
|
||||
exc.code,
|
||||
message,
|
||||
error_type=error_type,
|
||||
path=path,
|
||||
response_body=response_text,
|
||||
) from exc
|
||||
except AudioCppClientError:
|
||||
raise
|
||||
except (urllib.error.URLError, TimeoutError, socket.timeout, OSError) as exc:
|
||||
if self._is_timeout_error(exc):
|
||||
raise AudioCppTimeoutError(
|
||||
f"audio.cpp request to {path} timed out after {effective_timeout:.2f}s"
|
||||
) from exc
|
||||
reason = exc.reason if isinstance(exc, urllib.error.URLError) else exc
|
||||
raise AudioCppConnectionError(
|
||||
f"Could not connect to audio.cpp at {self.base_url}: {reason}"
|
||||
) from exc
|
||||
|
||||
def _request_json(
|
||||
self,
|
||||
method: str,
|
||||
path: str,
|
||||
*,
|
||||
payload: Optional[Mapping[str, Any]] = None,
|
||||
timeout: Optional[float] = None,
|
||||
) -> Dict[str, Any]:
|
||||
response_body, _ = self._request_bytes(method, path, payload=payload, timeout=timeout)
|
||||
try:
|
||||
decoded = json.loads(response_body.decode("utf-8"))
|
||||
except (UnicodeDecodeError, ValueError) as exc:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp returned invalid JSON for {path}: {exc}"
|
||||
) from exc
|
||||
if not isinstance(decoded, dict):
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp returned {type(decoded).__name__}, expected an object for {path}"
|
||||
)
|
||||
return decoded
|
||||
|
||||
def health(self, *, timeout: Optional[float] = None) -> Dict[str, Any]:
|
||||
return self._request_json(
|
||||
"GET",
|
||||
"/health",
|
||||
timeout=self.connect_timeout if timeout is None else timeout,
|
||||
)
|
||||
|
||||
def models(self, *, timeout: Optional[float] = None) -> list[Dict[str, Any]]:
|
||||
payload = self._request_json("GET", "/v1/models", timeout=timeout)
|
||||
models = payload.get("data")
|
||||
if not isinstance(models, list) or any(not isinstance(item, dict) for item in models):
|
||||
raise AudioCppProtocolError("audio.cpp /v1/models response is missing a valid data list")
|
||||
return list(models)
|
||||
|
||||
def voices(self, model_id: str, *, timeout: Optional[float] = None) -> list[str]:
|
||||
query = urllib.parse.urlencode({"model": str(model_id)})
|
||||
payload = self._request_json("GET", f"/v1/audio/voices?{query}", timeout=timeout)
|
||||
voices = payload.get("voices")
|
||||
if not isinstance(voices, list) or any(not isinstance(item, str) for item in voices):
|
||||
raise AudioCppProtocolError("audio.cpp voices response is missing a valid voices list")
|
||||
return list(voices)
|
||||
|
||||
def features(self, *, timeout: Optional[float] = None) -> set[str]:
|
||||
"""Read optional future feature metadata without assuming release-0.5.1 has it."""
|
||||
payload = self.health(timeout=timeout)
|
||||
raw = payload.get("features", payload.get("capabilities", []))
|
||||
if isinstance(raw, dict):
|
||||
return {str(name) for name, enabled in raw.items() if enabled}
|
||||
if isinstance(raw, list):
|
||||
return {str(item) for item in raw}
|
||||
return set()
|
||||
|
||||
def supports_feature(self, name: str, *, timeout: Optional[float] = None) -> bool:
|
||||
return str(name) in self.features(timeout=timeout)
|
||||
|
||||
def run_task(
|
||||
self,
|
||||
model_id: str,
|
||||
request: Mapping[str, Any],
|
||||
*,
|
||||
timeout: Optional[float] = None,
|
||||
) -> AudioCppTaskResult:
|
||||
if not isinstance(request, Mapping):
|
||||
raise TypeError("audio.cpp task request must be a mapping")
|
||||
payload = self._request_json(
|
||||
"POST",
|
||||
"/v1/tasks/run",
|
||||
payload={"model": str(model_id), "request": dict(request)},
|
||||
timeout=timeout,
|
||||
)
|
||||
|
||||
named_audio: Dict[str, AudioCppAudio] = {}
|
||||
raw_named = payload.get("named_audio_outputs", [])
|
||||
if raw_named is None:
|
||||
raw_named = []
|
||||
if not isinstance(raw_named, list):
|
||||
raise AudioCppProtocolError("audio.cpp named_audio_outputs must be a list")
|
||||
for index, item in enumerate(raw_named):
|
||||
if not isinstance(item, dict):
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp named_audio_outputs[{index}] must be an object"
|
||||
)
|
||||
output_id = item.get("id")
|
||||
if not isinstance(output_id, str) or not output_id:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp named_audio_outputs[{index}] is missing a non-empty id"
|
||||
)
|
||||
if output_id in named_audio:
|
||||
raise AudioCppProtocolError(f"audio.cpp returned duplicate named audio id: {output_id}")
|
||||
decoded = _decode_base64_wav(
|
||||
item.get("audio"), context=f"named_audio_outputs[{index}].audio"
|
||||
)
|
||||
self._validate_declared_audio_metadata(item, decoded, f"named_audio_outputs[{index}]")
|
||||
named_audio[output_id] = decoded
|
||||
|
||||
primary: Optional[AudioCppAudio] = None
|
||||
if "audio" in payload and payload.get("audio") is not None:
|
||||
primary = _decode_base64_wav(payload.get("audio"), context="audio")
|
||||
self._validate_declared_audio_metadata(payload, primary, "audio")
|
||||
elif len(named_audio) == 1:
|
||||
primary = next(iter(named_audio.values()))
|
||||
elif len(named_audio) > 1:
|
||||
raise AudioCppProtocolError(
|
||||
"audio.cpp task result contains multiple named audio outputs but no primary audio"
|
||||
)
|
||||
elif not named_audio and not any(
|
||||
key in payload for key in ("text", "segments", "speaker_turns", "words")
|
||||
):
|
||||
raise AudioCppProtocolError("audio.cpp task result did not contain task output")
|
||||
|
||||
return AudioCppTaskResult(
|
||||
waveform=primary.waveform if primary is not None else None,
|
||||
sample_rate=primary.sample_rate if primary is not None else None,
|
||||
channels=primary.channels if primary is not None else None,
|
||||
named_audio=named_audio,
|
||||
raw=payload,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_declared_audio_metadata(
|
||||
payload: Mapping[str, Any],
|
||||
decoded: AudioCppAudio,
|
||||
context: str,
|
||||
) -> None:
|
||||
declared_rate = payload.get("sample_rate")
|
||||
if declared_rate is not None and int(declared_rate) != decoded.sample_rate:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp {context} sample rate metadata ({declared_rate}) "
|
||||
f"does not match its WAV ({decoded.sample_rate})"
|
||||
)
|
||||
declared_channels = payload.get("channels")
|
||||
if declared_channels is not None and int(declared_channels) != decoded.channels:
|
||||
raise AudioCppProtocolError(
|
||||
f"audio.cpp {context} channel metadata ({declared_channels}) "
|
||||
f"does not match its WAV ({decoded.channels})"
|
||||
)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"AudioCppAudio",
|
||||
"AudioCppClient",
|
||||
"AudioCppClientError",
|
||||
"AudioCppConnectionError",
|
||||
"AudioCppHTTPError",
|
||||
"AudioCppProtocolError",
|
||||
"AudioCppTaskResult",
|
||||
"AudioCppTimeoutError",
|
||||
]
|
||||
@@ -0,0 +1,187 @@
|
||||
"""Resolve existing audio.cpp models without copying or cache migration."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Optional, Sequence, Union
|
||||
|
||||
from .catalog import AudioCppCatalog, PackageRecord, load_catalog
|
||||
from .settings import AudioCppSettings, get_settings_path, load_settings
|
||||
|
||||
|
||||
def _import_folder_paths():
|
||||
try:
|
||||
import folder_paths # type: ignore
|
||||
|
||||
return folder_paths
|
||||
except (ImportError, RuntimeError):
|
||||
return None
|
||||
|
||||
|
||||
def _registered_paths(folder_paths_module, key: str) -> list[Path]:
|
||||
if folder_paths_module is None:
|
||||
return []
|
||||
registry = getattr(folder_paths_module, "folder_names_and_paths", {})
|
||||
if key not in registry:
|
||||
return []
|
||||
try:
|
||||
if hasattr(folder_paths_module, "get_folder_paths"):
|
||||
values = folder_paths_module.get_folder_paths(key)
|
||||
else:
|
||||
values = registry[key][0]
|
||||
except (KeyError, TypeError, ValueError):
|
||||
return []
|
||||
return [Path(value).expanduser() for value in values if value]
|
||||
|
||||
|
||||
def _tts_paths(folder_paths_module) -> list[Path]:
|
||||
paths: list[Path] = []
|
||||
for key in ("TTS", "tts"):
|
||||
paths.extend(_registered_paths(folder_paths_module, key))
|
||||
if not paths and folder_paths_module is not None:
|
||||
models_dir = getattr(folder_paths_module, "models_dir", None)
|
||||
if models_dir:
|
||||
paths.append(Path(models_dir) / "TTS")
|
||||
return paths
|
||||
|
||||
|
||||
def default_managed_model_root(
|
||||
*,
|
||||
settings: Optional[AudioCppSettings] = None,
|
||||
folder_paths_module=None,
|
||||
) -> Path:
|
||||
current = settings or AudioCppSettings()
|
||||
if current.managed_model_root:
|
||||
return Path(current.managed_model_root).expanduser()
|
||||
module = folder_paths_module if folder_paths_module is not None else _import_folder_paths()
|
||||
tts_paths = _tts_paths(module)
|
||||
if tts_paths:
|
||||
return tts_paths[0] / "audio.cpp" / "models"
|
||||
# This is only a non-ComfyUI safety fallback; it remains direct storage, not
|
||||
# a Hugging Face cache.
|
||||
return get_settings_path(module).parent / "models"
|
||||
|
||||
|
||||
def _canonical(path: Path) -> str:
|
||||
return os.path.normcase(os.path.abspath(os.path.expanduser(str(path))))
|
||||
|
||||
|
||||
def resolve_model_roots(
|
||||
explicit_roots: Optional[Iterable[Union[str, Path]]] = None,
|
||||
*,
|
||||
settings: Optional[AudioCppSettings] = None,
|
||||
folder_paths_module=None,
|
||||
) -> list[Path]:
|
||||
"""Return search roots in priority order, with managed storage last."""
|
||||
|
||||
module = folder_paths_module if folder_paths_module is not None else _import_folder_paths()
|
||||
current = settings or load_settings(folder_paths_module=module)
|
||||
managed = default_managed_model_root(settings=current, folder_paths_module=module)
|
||||
candidates: list[Path] = []
|
||||
candidates.extend(Path(path).expanduser() for path in (explicit_roots or ()) if path)
|
||||
candidates.extend(Path(path).expanduser() for path in current.model_roots if path)
|
||||
candidates.extend(_registered_paths(module, "audio_cpp"))
|
||||
candidates.extend(path / "audio.cpp" / "models" for path in _tts_paths(module))
|
||||
|
||||
managed_key = _canonical(managed)
|
||||
seen: set[str] = set()
|
||||
result: list[Path] = []
|
||||
for candidate in candidates:
|
||||
key = _canonical(candidate)
|
||||
if key == managed_key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
result.append(candidate)
|
||||
result.append(managed)
|
||||
return result
|
||||
|
||||
|
||||
def package_install_path(package: PackageRecord, models_root: Union[str, Path]) -> Path:
|
||||
return Path(models_root) / Path(*package.target_directory.replace("\\", "/").split("/"))
|
||||
|
||||
|
||||
def package_is_complete(package: PackageRecord, package_directory: Union[str, Path]) -> bool:
|
||||
directory = Path(package_directory)
|
||||
return directory.is_dir() and all(
|
||||
(directory / path).is_file() and (directory / path).stat().st_size > 0
|
||||
for path in package.local_files
|
||||
)
|
||||
|
||||
|
||||
def find_installed_package(
|
||||
package: Union[str, PackageRecord],
|
||||
roots: Optional[Sequence[Union[str, Path]]] = None,
|
||||
*,
|
||||
catalog: Optional[AudioCppCatalog] = None,
|
||||
settings: Optional[AudioCppSettings] = None,
|
||||
folder_paths_module=None,
|
||||
) -> Optional[Path]:
|
||||
current_catalog = catalog or load_catalog()
|
||||
record = current_catalog.package(package) if isinstance(package, str) else package
|
||||
search_roots = (
|
||||
[Path(root) for root in roots]
|
||||
if roots is not None
|
||||
else resolve_model_roots(settings=settings, folder_paths_module=folder_paths_module)
|
||||
)
|
||||
for root in search_roots:
|
||||
candidate = package_install_path(record, root)
|
||||
if package_is_complete(record, candidate):
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ResolvedModel:
|
||||
package: PackageRecord
|
||||
path: Path
|
||||
root: Path
|
||||
|
||||
|
||||
def resolve_model(
|
||||
package: Union[str, PackageRecord],
|
||||
roots: Optional[Sequence[Union[str, Path]]] = None,
|
||||
*,
|
||||
catalog: Optional[AudioCppCatalog] = None,
|
||||
settings: Optional[AudioCppSettings] = None,
|
||||
folder_paths_module=None,
|
||||
) -> Optional[ResolvedModel]:
|
||||
"""Resolve one catalog package and retain the root that won precedence."""
|
||||
|
||||
current_catalog = catalog or load_catalog()
|
||||
record = current_catalog.package(package) if isinstance(package, str) else package
|
||||
search_roots = (
|
||||
[Path(root) for root in roots]
|
||||
if roots is not None
|
||||
else resolve_model_roots(settings=settings, folder_paths_module=folder_paths_module)
|
||||
)
|
||||
path = find_installed_package(record, search_roots, catalog=current_catalog)
|
||||
if path is None:
|
||||
return None
|
||||
for root in search_roots:
|
||||
if _canonical(package_install_path(record, root)) == _canonical(path):
|
||||
return ResolvedModel(package=record, path=path, root=root)
|
||||
return None
|
||||
|
||||
|
||||
def discover_installed_packages(
|
||||
roots: Optional[Sequence[Union[str, Path]]] = None,
|
||||
*,
|
||||
catalog: Optional[AudioCppCatalog] = None,
|
||||
settings: Optional[AudioCppSettings] = None,
|
||||
folder_paths_module=None,
|
||||
) -> dict[str, ResolvedModel]:
|
||||
current_catalog = catalog or load_catalog()
|
||||
found: dict[str, ResolvedModel] = {}
|
||||
for package in current_catalog.packages.values():
|
||||
resolved = resolve_model(
|
||||
package,
|
||||
roots,
|
||||
catalog=current_catalog,
|
||||
settings=settings,
|
||||
folder_paths_module=folder_paths_module,
|
||||
)
|
||||
if resolved is not None:
|
||||
found[package.id] = resolved
|
||||
return found
|
||||
@@ -0,0 +1,306 @@
|
||||
"""Direct, cache-free downloader for pinned audio.cpp model packages."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import uuid
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Callable, Mapping, Optional, Union
|
||||
from urllib.parse import quote
|
||||
|
||||
from .catalog import AudioCppCatalog, PackageRecord, load_catalog
|
||||
from .discovery import package_install_path, package_is_complete
|
||||
|
||||
|
||||
class AudioCppDownloadError(RuntimeError):
|
||||
"""Raised when a model package cannot be downloaded safely."""
|
||||
|
||||
|
||||
class IncompleteExistingPackageError(AudioCppDownloadError):
|
||||
"""Raised when a package target exists but is not complete."""
|
||||
|
||||
|
||||
ProgressCallback = Callable[[str, int, Optional[int]], None]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DownloadResult:
|
||||
package: PackageRecord
|
||||
path: Path
|
||||
downloaded_files: tuple[Path, ...]
|
||||
bytes_downloaded: int
|
||||
already_present: bool = False
|
||||
|
||||
|
||||
def resolve_hf_token(explicit_token: Optional[str] = None) -> Optional[str]:
|
||||
if explicit_token:
|
||||
return explicit_token.strip() or None
|
||||
for name in ("HF_TOKEN", "HUGGING_FACE_HUB_TOKEN"):
|
||||
value = os.environ.get(name, "").strip()
|
||||
if value:
|
||||
return value
|
||||
try:
|
||||
from huggingface_hub import get_token
|
||||
|
||||
value = get_token()
|
||||
if value:
|
||||
return value.strip()
|
||||
except (ImportError, OSError, RuntimeError):
|
||||
pass
|
||||
token_path = Path.home() / ".cache" / "huggingface" / "token"
|
||||
try:
|
||||
value = token_path.read_text(encoding="utf-8").strip()
|
||||
return value or None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
|
||||
def huggingface_resolve_url(repo: str, revision: str, remote_path: str) -> str:
|
||||
return (
|
||||
"https://huggingface.co/"
|
||||
f"{quote(repo, safe='/')}/resolve/{quote(revision, safe='')}/"
|
||||
f"{quote(remote_path.replace(chr(92), '/'), safe='/')}"
|
||||
)
|
||||
|
||||
|
||||
def package_download_size(
|
||||
package: Union[str, PackageRecord],
|
||||
*,
|
||||
catalog: Optional[AudioCppCatalog] = None,
|
||||
token: Optional[str] = None,
|
||||
timeout: int = 30,
|
||||
opener=None,
|
||||
) -> Optional[int]:
|
||||
"""Return the declared HTTP size of every package file when available."""
|
||||
|
||||
current_catalog = catalog or load_catalog()
|
||||
record = current_catalog.package(package) if isinstance(package, str) else package
|
||||
headers = {"User-Agent": "TTS-Audio-Suite/audio.cpp-model-installer"}
|
||||
hf_token = resolve_hf_token(token)
|
||||
if hf_token:
|
||||
headers["Authorization"] = f"Bearer {hf_token}"
|
||||
open_request = opener or urllib.request.urlopen
|
||||
total = 0
|
||||
try:
|
||||
for remote_path in record.files:
|
||||
request = urllib.request.Request(
|
||||
huggingface_resolve_url(record.repo, record.revision, remote_path),
|
||||
headers=headers,
|
||||
method="HEAD",
|
||||
)
|
||||
response = open_request(request, timeout=timeout)
|
||||
try:
|
||||
length = response.headers.get("Content-Length") if response.headers else None
|
||||
if not length:
|
||||
return None
|
||||
total += int(length)
|
||||
finally:
|
||||
response.close()
|
||||
except (OSError, TypeError, ValueError, urllib.error.URLError):
|
||||
return None
|
||||
return total or None
|
||||
|
||||
|
||||
def download_url_to_path(
|
||||
url: str,
|
||||
destination: Union[str, Path],
|
||||
*,
|
||||
headers: Optional[Mapping[str, str]] = None,
|
||||
timeout: int = 300,
|
||||
opener=None,
|
||||
progress: Optional[ProgressCallback] = None,
|
||||
progress_label: str = "download",
|
||||
) -> int:
|
||||
"""Stream a URL to a new file and validate HTTP Content-Length."""
|
||||
|
||||
destination_path = Path(destination)
|
||||
request = urllib.request.Request(url, headers=dict(headers or {}))
|
||||
open_request = opener or urllib.request.urlopen
|
||||
response = None
|
||||
downloaded = 0
|
||||
try:
|
||||
response = open_request(request, timeout=timeout)
|
||||
status = getattr(response, "status", None) or getattr(response, "code", None)
|
||||
if status is not None and int(status) >= 400:
|
||||
raise AudioCppDownloadError(f"HTTP {status} while downloading {url}")
|
||||
content_length_value = response.headers.get("Content-Length") if response.headers else None
|
||||
expected = int(content_length_value) if content_length_value else None
|
||||
destination_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with destination_path.open("xb") as handle:
|
||||
while True:
|
||||
chunk = response.read(1024 * 1024)
|
||||
if not chunk:
|
||||
break
|
||||
handle.write(chunk)
|
||||
downloaded += len(chunk)
|
||||
if progress is not None:
|
||||
progress(progress_label, downloaded, expected)
|
||||
if expected is not None and downloaded != expected:
|
||||
raise AudioCppDownloadError(
|
||||
f"Incomplete download for {progress_label}: expected {expected} bytes, got {downloaded}"
|
||||
)
|
||||
return downloaded
|
||||
except urllib.error.HTTPError as exc:
|
||||
if exc.code in (401, 403):
|
||||
raise AudioCppDownloadError(
|
||||
f"Hugging Face denied {progress_label} (HTTP {exc.code}). "
|
||||
"Accept any model license and configure HF_TOKEN."
|
||||
) from exc
|
||||
raise AudioCppDownloadError(f"HTTP {exc.code} while downloading {progress_label}") from exc
|
||||
except urllib.error.URLError as exc:
|
||||
raise AudioCppDownloadError(f"Network error downloading {progress_label}: {exc.reason}") from exc
|
||||
except OSError as exc:
|
||||
raise AudioCppDownloadError(f"Cannot write {destination_path}: {exc}") from exc
|
||||
finally:
|
||||
if response is not None:
|
||||
try:
|
||||
response.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _remove_path(path: Path) -> None:
|
||||
if path.is_symlink() or path.is_file():
|
||||
path.unlink(missing_ok=True)
|
||||
elif path.is_dir():
|
||||
shutil.rmtree(path)
|
||||
|
||||
|
||||
def _link_or_copy(source: str, destination: str) -> str:
|
||||
"""Hard-link existing model files into staging, copying only if necessary."""
|
||||
|
||||
try:
|
||||
os.link(source, destination)
|
||||
return destination
|
||||
except OSError:
|
||||
return shutil.copy2(source, destination)
|
||||
|
||||
|
||||
def _merge_existing_target(target: Path, staging: Path) -> None:
|
||||
"""Preserve other precision packages that share this target directory."""
|
||||
|
||||
if target.is_symlink() or not target.is_dir():
|
||||
raise IncompleteExistingPackageError(
|
||||
f"Model target cannot be safely merged because it is not a regular directory: {target}"
|
||||
)
|
||||
shutil.copytree(
|
||||
target,
|
||||
staging,
|
||||
dirs_exist_ok=True,
|
||||
copy_function=_link_or_copy,
|
||||
symlinks=False,
|
||||
)
|
||||
|
||||
|
||||
def _publish_directory(staging: Path, target: Path, replace_existing: bool) -> None:
|
||||
# PocketTTS targets are nested (for example PocketTTS-GGUF/english).
|
||||
# The parent must exist before Windows can rename the staged directory.
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
if not target.exists() and not target.is_symlink():
|
||||
staging.rename(target)
|
||||
return
|
||||
if not replace_existing:
|
||||
raise IncompleteExistingPackageError(
|
||||
f"Model target already exists but is incomplete: {target}. "
|
||||
"Choose overwrite explicitly or repair it manually."
|
||||
)
|
||||
backup = target.with_name(f".{target.name}.{uuid.uuid4().hex}.backup")
|
||||
target.rename(backup)
|
||||
try:
|
||||
staging.rename(target)
|
||||
except BaseException:
|
||||
if not target.exists() and backup.exists():
|
||||
backup.rename(target)
|
||||
raise
|
||||
_remove_path(backup)
|
||||
|
||||
|
||||
def install_package(
|
||||
package: Union[str, PackageRecord],
|
||||
models_root: Union[str, Path],
|
||||
*,
|
||||
catalog: Optional[AudioCppCatalog] = None,
|
||||
overwrite: bool = False,
|
||||
token: Optional[str] = None,
|
||||
timeout: int = 300,
|
||||
opener=None,
|
||||
progress: Optional[ProgressCallback] = None,
|
||||
) -> DownloadResult:
|
||||
"""Download every required package file, then atomically publish it."""
|
||||
|
||||
current_catalog = catalog or load_catalog()
|
||||
record = current_catalog.package(package) if isinstance(package, str) else package
|
||||
if record.download.get("kind") != "huggingface_snapshot":
|
||||
raise AudioCppDownloadError(
|
||||
f"Unsupported download kind for {record.id}: {record.download.get('kind')!r}"
|
||||
)
|
||||
if not record.repo:
|
||||
raise AudioCppDownloadError(f"Package {record.id!r} has no Hugging Face repository")
|
||||
|
||||
root = Path(models_root).expanduser()
|
||||
target = package_install_path(record, root)
|
||||
if package_is_complete(record, target) and not overwrite:
|
||||
return DownloadResult(record, target, (), 0, already_present=True)
|
||||
target_exists = target.exists() or target.is_symlink()
|
||||
if target_exists:
|
||||
if target.is_symlink() or not target.is_dir():
|
||||
raise IncompleteExistingPackageError(
|
||||
f"Model target cannot be safely merged because it is not a regular directory: {target}"
|
||||
)
|
||||
requested_files_exist = any(
|
||||
(target / path).exists() or (target / path).is_symlink()
|
||||
for path in record.local_files
|
||||
)
|
||||
if requested_files_exist and not overwrite:
|
||||
raise IncompleteExistingPackageError(
|
||||
f"Model target contains an incomplete package {record.id!r}: {target}. "
|
||||
"Choose overwrite explicitly or repair it manually."
|
||||
)
|
||||
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
staging = Path(tempfile.mkdtemp(prefix=f".{target.name}.", suffix=".staging", dir=target.parent))
|
||||
downloaded_files: list[Path] = []
|
||||
downloaded_bytes = 0
|
||||
hf_token = resolve_hf_token(token)
|
||||
headers = {"User-Agent": "TTS-Audio-Suite/audio.cpp-model-installer"}
|
||||
if hf_token:
|
||||
headers["Authorization"] = f"Bearer {hf_token}"
|
||||
try:
|
||||
if target_exists:
|
||||
_merge_existing_target(target, staging)
|
||||
for remote_path, local_path in zip(record.files, record.local_files):
|
||||
destination = staging / local_path
|
||||
# The merge may have hard-linked an older copy of the requested
|
||||
# precision. Remove that link before creating the replacement.
|
||||
destination.unlink(missing_ok=True)
|
||||
url = huggingface_resolve_url(record.repo, record.revision, remote_path)
|
||||
downloaded_bytes += download_url_to_path(
|
||||
url,
|
||||
destination,
|
||||
headers=headers,
|
||||
timeout=timeout,
|
||||
opener=opener,
|
||||
progress=progress,
|
||||
progress_label=remote_path,
|
||||
)
|
||||
downloaded_files.append(local_path)
|
||||
if not package_is_complete(record, staging):
|
||||
missing = [str(path) for path in record.local_files if not (staging / path).is_file()]
|
||||
raise AudioCppDownloadError(
|
||||
f"Package {record.id!r} staging validation failed; missing: {missing}"
|
||||
)
|
||||
_publish_directory(staging, target, replace_existing=target_exists)
|
||||
return DownloadResult(
|
||||
package=record,
|
||||
path=target,
|
||||
downloaded_files=tuple(downloaded_files),
|
||||
bytes_downloaded=downloaded_bytes,
|
||||
)
|
||||
except BaseException:
|
||||
_remove_path(staging)
|
||||
raise
|
||||
@@ -0,0 +1,232 @@
|
||||
schema_version: 1
|
||||
release: release-0.5.1
|
||||
|
||||
# Verified from Hugging Face file metadata on 2026-08-13. These are offline UI
|
||||
# estimates; live Content-Length values remain authoritative during downloads.
|
||||
package_sizes:
|
||||
citrinet_asr_q8_0: 40574432
|
||||
chatterbox_f16: 3744360386
|
||||
chatterbox_q8_0: 2088393668
|
||||
chatterbox_safetensors: 3191859618
|
||||
confucius4_tts_orig: 8192757760
|
||||
dramabox_q8_0: 18942803808
|
||||
fish_audio_s2_pro_bf16: 10229278080
|
||||
fish_audio_s2_pro_q8_0: 6317911232
|
||||
fun_asr_nano_2512_f16: 1675708832
|
||||
fun_asr_nano_2512_q8_0: 1045334432
|
||||
fun_asr_nano_2512_safetensors: 1671205126
|
||||
glm_tts_q8_0: 5143813728
|
||||
higgs_audio_tts_4b_bf16: 8501587648
|
||||
higgs_audio_tts_4b_q8_0: 5095354048
|
||||
higgs_audio_stt_f16: 5367453248
|
||||
higgs_audio_stt_q8_0: 3158310848
|
||||
hviske_asr_q8_0: 2436359808
|
||||
hviske_asr_safetensors: 4132293306
|
||||
index_tts2_f16: 4646898304
|
||||
index_tts2_orig: 8084552000
|
||||
index_tts2_q8_0: 3633888608
|
||||
index_tts2_safetensors: 3484956803
|
||||
inflect_micro_v2_orig: 72082176
|
||||
irodori_tts_500m_v3_f16: 1254813120
|
||||
irodori_tts_500m_v3_q8_0: 1093739584
|
||||
irodori_tts_600m_v3_voicedesign_f16: 1463787680
|
||||
irodori_tts_600m_v3_voicedesign_q8_0: 1272140064
|
||||
irodori_tts_v4_small_f16: 1762148352
|
||||
irodori_tts_v4_small_q8_0: 1368991360
|
||||
kroko_asr_community_q8_0: 167756928
|
||||
miotts_1_7b_bf16: 3518532512
|
||||
miotts_1_7b_orig: 3518532512
|
||||
miotts_1_7b_q8_0: 2197326752
|
||||
moss_tts_local_v1_5_bf16: 13367797280
|
||||
moss_tts_local_v1_5_q8_0: 7512220768
|
||||
moss_tts_nano_100m_bf16: 332423040
|
||||
moss_tts_nano_100m_q8_0: 193337984
|
||||
nemotron_asr_f16: 1277710880
|
||||
nemotron_asr_q8_0: 930620256
|
||||
nemotron_asr_safetensors: 2552818890
|
||||
omnivoice_bf16: 1639548640
|
||||
omnivoice_f16: 1639548768
|
||||
omnivoice_q8_0: 1350288416
|
||||
omnivoice_safetensors: 2461770336
|
||||
outetts_1_0_1b_q8_0: 3029895456
|
||||
parakeet_tdt_f16: 1255384320
|
||||
parakeet_tdt_q8_0: 915733744
|
||||
pocket_tts_english_bf16: 219096064
|
||||
pocket_tts_english_q8_0: 127856704
|
||||
pocket_tts_english_safetensors: 385107895
|
||||
pocket_tts_german_bf16: 219096544
|
||||
pocket_tts_german_q8_0: 127857184
|
||||
pocket_tts_italian_bf16: 219096800
|
||||
pocket_tts_italian_q8_0: 127857440
|
||||
pocket_tts_portuguese_bf16: 219097728
|
||||
pocket_tts_portuguese_q8_0: 127858368
|
||||
pocket_tts_spanish_bf16: 219097600
|
||||
pocket_tts_spanish_q8_0: 127858240
|
||||
qwen3_tts_0_6b_base_safetensors: 2511651783
|
||||
qwen3_tts_1_7b_base_bf16: 4203158464
|
||||
qwen3_tts_1_7b_base_orig: 4544273280
|
||||
qwen3_tts_1_7b_base_q8_0: 2695175104
|
||||
qwen3_tts_1_7b_base_safetensors: 4539721255
|
||||
qwen3_tts_1_7b_customvoice_bf16: 4179144352
|
||||
qwen3_tts_1_7b_customvoice_q8_0: 2817044064
|
||||
qwen3_tts_1_7b_voicedesign_bf16: 4179089248
|
||||
qwen3_tts_1_7b_voicedesign_q8_0: 2816988960
|
||||
qwen3_asr_0_6b_f16: 1880642016
|
||||
qwen3_asr_0_6b_q8_0: 1151272416
|
||||
qwen3_asr_0_6b_safetensors: 1876110856
|
||||
qwen3_asr_1_7b_f16: 4087653248
|
||||
qwen3_asr_1_7b_q8_0: 2473010048
|
||||
qwen3_asr_1_7b_safetensors: 4087626782
|
||||
seed_vc_mlx_f16: 3629186560
|
||||
seed_vc_mlx_orig: 7003311936
|
||||
seed_vc_mlx_q8_0: 3120809248
|
||||
seed_vc_mlx_safetensors: 712430273
|
||||
supertonic_3_f16: 312784196
|
||||
supertonic_3_orig: 454072836
|
||||
supertonic_3_q8_0: 454072836
|
||||
supertonic_3_safetensors: 397279577
|
||||
vevo2_f16: 5021112128
|
||||
vevo2_orig: 7449615488
|
||||
vevo2_q8_0: 3242124800
|
||||
vibevoice_1_5b_bf16: 5420021858
|
||||
vibevoice_1_5b_q8_0: 3224701538
|
||||
vibevoice_asr_f16: 17361090304
|
||||
vibevoice_asr_q8_0: 9858644224
|
||||
vietneu_tts_v3_turbo_q8_0: 170482368
|
||||
voxcpm2_bf16: 4772288288
|
||||
voxcpm2_orig: 4960716192
|
||||
voxcpm2_q8_0: 2955000480
|
||||
voxcpm2_safetensors: 4583766759
|
||||
voxtral_realtime_bf16: 8874402784
|
||||
voxtral_realtime_q4_k: 3097662432
|
||||
voxtral_realtime_q8_0: 5104567264
|
||||
|
||||
# Runtime dependencies omitted by upstream TTS package declarations. These are
|
||||
# installed independently and passed to audio.cpp through absolute session paths.
|
||||
package_dependencies:
|
||||
miotts_1_7b_q8_0: &miotts_codec_dependency
|
||||
- id: miocodec_q8_0
|
||||
family: miocodec
|
||||
display_name: "MioCodec 25Hz 44.1kHz v2 Q8_0 GGUF"
|
||||
target_directory: "MioCodec-25Hz-44.1kHz-v2-GGUF"
|
||||
format: gguf
|
||||
precision: q8_0
|
||||
files: ["MioCodec-25Hz-44.1kHz-v2-GGUF/miocodec-25hz-44khz-v2-q8_0.gguf"]
|
||||
strip_prefix: "MioCodec-25Hz-44.1kHz-v2-GGUF"
|
||||
repo: "audio-cpp/audio.cpp-gguf"
|
||||
revision: main
|
||||
estimated_download_bytes: 299066464
|
||||
session_option: "miotts.codec_model_path"
|
||||
miotts_1_7b_bf16: *miotts_codec_dependency
|
||||
miotts_1_7b_orig: *miotts_codec_dependency
|
||||
|
||||
# Suite-owned UI and validation metadata. Download/package truth remains in
|
||||
# model_specs/*.json, which are unmodified snapshots of audio.cpp.
|
||||
defaults:
|
||||
suite_support: supported
|
||||
suite_tasks: [tts, srt, character_switching]
|
||||
reference_audio: none
|
||||
reference_transcript: none
|
||||
built_in_voices: false
|
||||
voice_design: false
|
||||
inline_controls: false
|
||||
native_multi_speaker: {supported: false, max_speakers: 1, suite_status: unavailable}
|
||||
asr_features: {diarization: none, timing: none}
|
||||
|
||||
families:
|
||||
citrinet_asr:
|
||||
suite_tasks: [asr]
|
||||
summary: "Compact English transcription using a Citrinet CTC model."
|
||||
chatterbox:
|
||||
reference_audio: required
|
||||
suite_tasks: [tts, srt, character_switching, voice_conversion]
|
||||
summary: "Voice cloning and reference-targeted voice conversion."
|
||||
confucius4_tts: {reference_audio: required, summary: "Voice cloning from reference audio."}
|
||||
dramabox: {reference_audio: optional, summary: "TTS and optional voice cloning."}
|
||||
fish_audio: {reference_audio: optional, summary: "Reference-conditioned TTS."}
|
||||
fun_asr_nano:
|
||||
suite_tasks: [asr]
|
||||
summary: "Multilingual speech recognition for Chinese, English, and Japanese."
|
||||
glm_tts:
|
||||
reference_audio: required
|
||||
reference_transcript: required
|
||||
summary: "Zero-shot cloning; reference audio and its transcript are required."
|
||||
higgs_audio_tts:
|
||||
reference_audio: optional
|
||||
reference_transcript: optional
|
||||
inline_controls: true
|
||||
summary: "Expressive multilingual TTS, cloning, and inline style/sound controls."
|
||||
higgs_audio_stt:
|
||||
suite_tasks: [asr]
|
||||
summary: "English transcription using the Higgs Audio v3 speech model."
|
||||
hviske_asr:
|
||||
suite_tasks: [asr]
|
||||
summary: "Multilingual transcription across 16 languages."
|
||||
index_tts2: {reference_audio: optional, summary: "TTS with optional voice cloning."}
|
||||
inflect_v2: {summary: "Reference-free expressive TTS."}
|
||||
irodori_tts:
|
||||
reference_audio: optional
|
||||
voice_design: true
|
||||
summary: "TTS, optional cloning, and voice design."
|
||||
kroko_asr:
|
||||
suite_tasks: [asr]
|
||||
asr_features: {timing: native_word}
|
||||
summary: "Small multilingual community ASR model."
|
||||
miotts: {reference_audio: optional, summary: "Reference-conditioned TTS."}
|
||||
moss_tts_local: {reference_audio: optional, summary: "TTS with optional voice cloning."}
|
||||
moss_tts_nano: {reference_audio: optional, summary: "Compact TTS with optional cloning."}
|
||||
nemotron_asr:
|
||||
suite_tasks: [asr]
|
||||
asr_features: {timing: native_word}
|
||||
summary: "Streaming-capable multilingual NVIDIA Nemotron transcription."
|
||||
omnivoice:
|
||||
reference_audio: optional
|
||||
reference_transcript: optional
|
||||
voice_design: true
|
||||
summary: "Reference cloning or instruction-based voice design."
|
||||
outetts:
|
||||
reference_audio: optional
|
||||
reference_transcript: optional
|
||||
summary: "Multilingual TTS; a transcript improves reference cloning."
|
||||
parakeet_tdt:
|
||||
suite_tasks: [asr]
|
||||
asr_features: {timing: native_word}
|
||||
summary: "Fast multilingual Parakeet-TDT transcription."
|
||||
pocket_tts: {reference_audio: optional, summary: "Small CPU-friendly TTS and cloning."}
|
||||
qwen3_tts:
|
||||
reference_audio: optional
|
||||
voice_design: true
|
||||
summary: "TTS with package-dependent cloning or voice design."
|
||||
qwen3_asr:
|
||||
suite_tasks: [asr]
|
||||
asr_features: {timing: optional_forced_aligner}
|
||||
summary: "Broad multilingual Qwen3 transcription; word timestamps require an optional forced aligner."
|
||||
seed_vc:
|
||||
suite_tasks: [voice_conversion]
|
||||
reference_audio: required
|
||||
summary: "Reference-targeted voice conversion through Seed-VC."
|
||||
supertonic:
|
||||
built_in_voices: true
|
||||
summary: "Fast multilingual TTS using built-in voices."
|
||||
vevo2:
|
||||
reference_audio: optional
|
||||
suite_tasks: [tts, srt, character_switching, voice_conversion]
|
||||
summary: "TTS and reference-targeted voice conversion; S2S/SVC remain advanced upstream routes."
|
||||
vibevoice:
|
||||
reference_audio: required_per_speaker
|
||||
native_multi_speaker: {supported: true, max_speakers: 4, suite_status: partial}
|
||||
summary: "Long-form dialogue for up to four speakers. Native single-request mode is not wired yet."
|
||||
vibevoice_asr:
|
||||
suite_tasks: [asr, diarization]
|
||||
asr_features: {diarization: native, timing: native_segment}
|
||||
summary: "Multilingual transcription with native timestamped speaker turns."
|
||||
vietneu_tts:
|
||||
voice_design: true
|
||||
summary: "Vietnamese TTS and voice design."
|
||||
voxcpm2:
|
||||
reference_audio: optional
|
||||
voice_design: true
|
||||
summary: "TTS with reference conditioning and voice design."
|
||||
voxtral_realtime:
|
||||
suite_tasks: [asr]
|
||||
summary: "Multilingual realtime-oriented Voxtral transcription."
|
||||
@@ -0,0 +1,287 @@
|
||||
"""ComfyUI model-management proxy for suite-owned audio.cpp processes."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
import weakref
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Any, Dict, Mapping, Optional
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from .session import AudioCppSession
|
||||
|
||||
|
||||
def _first(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
|
||||
for key in keys:
|
||||
if key in config and config[key] is not None:
|
||||
return config[key]
|
||||
return default
|
||||
|
||||
|
||||
def _warn(message: str, exc: Optional[BaseException] = None) -> None:
|
||||
text = f"WARNING: {message}"
|
||||
if exc is not None:
|
||||
text += f": {exc}"
|
||||
encoding = getattr(sys.stderr, "encoding", None) or "ascii"
|
||||
try:
|
||||
text = text.encode(encoding, errors="replace").decode(encoding, errors="replace")
|
||||
except LookupError:
|
||||
text = text.encode("ascii", errors="replace").decode("ascii")
|
||||
print(text, file=sys.stderr)
|
||||
|
||||
|
||||
class AudioCppRuntimeProxy:
|
||||
"""ComfyUI model-like resource representing one owned native server."""
|
||||
|
||||
def __init__(self, session: "AudioCppSession") -> None:
|
||||
self._session_ref = weakref.ref(session)
|
||||
self.model = self
|
||||
self.processor = self
|
||||
self.parent = None
|
||||
self.currently_used = True
|
||||
self.model_options: Dict[str, Any] = {}
|
||||
self.model_keys = set()
|
||||
self.offload_device = self._torch_device("cpu", 0)
|
||||
backend = str(session.config.get("backend", "cuda")).lower()
|
||||
device_index = self._device_index(
|
||||
session.config.get("device_index", session.config.get("device", 0))
|
||||
)
|
||||
self.load_device = self._lifecycle_device(backend, device_index)
|
||||
self.current_device = self.load_device
|
||||
self._estimated_memory_size = self._memory_estimate(session.config)
|
||||
self._loaded_model = None
|
||||
self._model_management = None
|
||||
self._registration_lock = threading.RLock()
|
||||
|
||||
@staticmethod
|
||||
def _device_index(value: Any) -> int:
|
||||
text = str(value or 0).lower()
|
||||
if ":" in text:
|
||||
text = text.rsplit(":", 1)[-1]
|
||||
try:
|
||||
return max(0, int(text))
|
||||
except ValueError:
|
||||
return 0
|
||||
|
||||
@staticmethod
|
||||
def _torch_device(kind: str, index: int):
|
||||
try:
|
||||
import torch
|
||||
|
||||
return torch.device(kind, index) if kind == "cuda" else torch.device(kind)
|
||||
except Exception:
|
||||
return f"{kind}:{index}" if kind == "cuda" else kind
|
||||
|
||||
@classmethod
|
||||
def _lifecycle_device(cls, backend: str, index: int):
|
||||
if backend == "cpu":
|
||||
return cls._torch_device("cpu", 0)
|
||||
try:
|
||||
import comfy.model_management as model_management
|
||||
|
||||
device = model_management.get_torch_device()
|
||||
if device is not None:
|
||||
return device
|
||||
except (ImportError, AttributeError, RuntimeError):
|
||||
pass
|
||||
if backend == "metal":
|
||||
return cls._torch_device("mps", 0)
|
||||
return cls._torch_device("cuda", index)
|
||||
|
||||
@staticmethod
|
||||
def _memory_estimate(config: Mapping[str, Any]) -> int:
|
||||
explicit = _first(config, "estimated_memory_bytes", "model_memory_bytes")
|
||||
if explicit is not None:
|
||||
return max(1, int(explicit))
|
||||
gigabytes = _first(config, "estimated_vram_gb", "model_memory_gb")
|
||||
if gigabytes is not None:
|
||||
return max(1, int(float(gigabytes) * 1024**3))
|
||||
|
||||
raw_path = _first(config, "model_path", "package_path", "gguf_path")
|
||||
if raw_path:
|
||||
try:
|
||||
model_path = Path(
|
||||
os.path.expandvars(os.path.expanduser(str(raw_path)))
|
||||
).resolve()
|
||||
if model_path.is_file():
|
||||
return max(1, model_path.stat().st_size)
|
||||
if model_path.is_dir():
|
||||
total = 0
|
||||
for candidate in model_path.rglob("*"):
|
||||
try:
|
||||
if candidate.is_file():
|
||||
total += candidate.stat().st_size
|
||||
except OSError:
|
||||
continue
|
||||
return max(1, total)
|
||||
except OSError:
|
||||
pass
|
||||
# A zero-size entry is ignored by parts of ComfyUI's unload ordering.
|
||||
return 1
|
||||
|
||||
def _session(self) -> Optional["AudioCppSession"]:
|
||||
return self._session_ref()
|
||||
|
||||
def register(self) -> bool:
|
||||
"""Register once with ComfyUI after an owned process becomes live."""
|
||||
with self._registration_lock:
|
||||
if self._loaded_model is not None:
|
||||
current = getattr(self._model_management, "current_loaded_models", None)
|
||||
if isinstance(current, list) and self._loaded_model in current:
|
||||
return True
|
||||
self._loaded_model = None
|
||||
self._model_management = None
|
||||
try:
|
||||
import comfy.model_management as model_management
|
||||
except ImportError:
|
||||
return False
|
||||
try:
|
||||
loaded_model_type = getattr(model_management, "LoadedModel", None)
|
||||
current_models = getattr(model_management, "current_loaded_models", None)
|
||||
if not callable(loaded_model_type) or not isinstance(current_models, list):
|
||||
return False
|
||||
loaded_model = loaded_model_type(self)
|
||||
loaded_model.real_model = weakref.ref(self)
|
||||
cleanup = getattr(model_management, "cleanup_models", None)
|
||||
loaded_model.model_finalizer = weakref.finalize(
|
||||
self, cleanup if callable(cleanup) else lambda: None
|
||||
)
|
||||
loaded_model._tts_wrapper_ref = self
|
||||
current_models.insert(0, loaded_model)
|
||||
self._loaded_model = loaded_model
|
||||
self._model_management = model_management
|
||||
return True
|
||||
except Exception as exc:
|
||||
_warn("Failed to register audio.cpp runtime with ComfyUI", exc)
|
||||
return False
|
||||
|
||||
def unregister(self) -> None:
|
||||
with self._registration_lock:
|
||||
loaded_model = self._loaded_model
|
||||
model_management = self._model_management
|
||||
self._loaded_model = None
|
||||
self._model_management = None
|
||||
if loaded_model is None or model_management is None:
|
||||
return
|
||||
current_models = getattr(model_management, "current_loaded_models", None)
|
||||
if not isinstance(current_models, list):
|
||||
return
|
||||
try:
|
||||
for candidate in list(current_models):
|
||||
if candidate is loaded_model or getattr(candidate, "_tts_wrapper_ref", None) is self:
|
||||
current_models.remove(candidate)
|
||||
except Exception as exc:
|
||||
_warn("Failed to unregister audio.cpp runtime from ComfyUI", exc)
|
||||
|
||||
def to(self, device):
|
||||
self.current_device = device
|
||||
return self
|
||||
|
||||
def eval(self):
|
||||
return self
|
||||
|
||||
def model_size(self) -> int:
|
||||
return self._estimated_memory_size
|
||||
|
||||
def loaded_size(self) -> int:
|
||||
session = self._session()
|
||||
return self._estimated_memory_size if session is not None and session.running else 0
|
||||
|
||||
def model_memory(self) -> int:
|
||||
return self.model_size()
|
||||
|
||||
def get_ram_usage(self) -> int:
|
||||
return self._estimated_memory_size
|
||||
|
||||
def model_offloaded_memory(self) -> int:
|
||||
return max(0, self.model_size() - self.loaded_size())
|
||||
|
||||
def model_mmap_residency(self, free: bool = False) -> tuple[int, int]:
|
||||
return 0, self._estimated_memory_size
|
||||
|
||||
def pinned_memory_size(self) -> int:
|
||||
return 0
|
||||
|
||||
def lowvram_patch_counter(self) -> int:
|
||||
return 0
|
||||
|
||||
def model_dtype(self):
|
||||
try:
|
||||
import torch
|
||||
|
||||
return torch.float32
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def current_loaded_device(self):
|
||||
return self.current_device
|
||||
|
||||
def model_patches_models(self):
|
||||
return ()
|
||||
|
||||
def model_patches_to(self, target) -> None:
|
||||
try:
|
||||
import torch
|
||||
|
||||
if isinstance(target, torch.device):
|
||||
self.current_device = target
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def is_dynamic(self) -> bool:
|
||||
return False
|
||||
|
||||
def is_clone(self, other) -> bool:
|
||||
return other is self
|
||||
|
||||
def clone_has_same_weights(self, other) -> bool:
|
||||
return other is self
|
||||
|
||||
def partially_load(self, device, extra_memory, force_patch_weights=False) -> int:
|
||||
self.current_device = device
|
||||
return 0
|
||||
|
||||
def partially_unload(self, device, memory_to_free) -> int:
|
||||
# audio.cpp 0.5.1 cannot release part of a session. Claiming memory here
|
||||
# would make ComfyUI believe VRAM was freed while the server still owns it.
|
||||
return 0
|
||||
|
||||
def partially_unload_ram(self, ram_to_unload) -> int:
|
||||
return 0
|
||||
|
||||
def patch_model(
|
||||
self,
|
||||
device_to=None,
|
||||
lowvram_model_memory=0,
|
||||
load_weights=True,
|
||||
force_patch_weights=False,
|
||||
):
|
||||
if device_to is not None:
|
||||
self.current_device = device_to
|
||||
return self.model
|
||||
|
||||
def unpatch_model(self, device_to=None, unpatch_weights=True):
|
||||
if device_to is not None:
|
||||
self.current_device = device_to
|
||||
session = self._session()
|
||||
if session is not None:
|
||||
# LoadedModel removes its list entry after this callback returns.
|
||||
session._stop_owned_runtime(unregister=False)
|
||||
return self.model
|
||||
|
||||
def model_unload(self, memory_to_free=None, unpatch_weights=True) -> bool:
|
||||
self.unpatch_model(self.offload_device, unpatch_weights=unpatch_weights)
|
||||
return True
|
||||
|
||||
def detach(self, unpatch_weights=True):
|
||||
return self.unpatch_model(self.offload_device, unpatch_weights=unpatch_weights)
|
||||
|
||||
def cleanup(self) -> None:
|
||||
session = self._session()
|
||||
if session is not None:
|
||||
session._stop_owned_runtime(unregister=True)
|
||||
|
||||
|
||||
__all__ = ["AudioCppRuntimeProxy"]
|
||||
@@ -0,0 +1,168 @@
|
||||
{
|
||||
"family": "chatterbox",
|
||||
"display_name": "Chatterbox",
|
||||
"description": "Open-source Chatterbox family for expressive TTS and voice conversion, with emotion exaggeration control, fast generation, zero-shot voice cloning, and an integrated multilingual TTS path.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone",
|
||||
"vc"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"ar",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"en",
|
||||
"es",
|
||||
"fi",
|
||||
"fr",
|
||||
"hi",
|
||||
"it",
|
||||
"ko",
|
||||
"ms",
|
||||
"nl",
|
||||
"no",
|
||||
"pl",
|
||||
"pt",
|
||||
"sv",
|
||||
"sw",
|
||||
"tr"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"vc": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "chatterbox_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"VC",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "chatterbox_q8_0",
|
||||
"display_name": "Chatterbox Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Chatterbox-GGUF",
|
||||
"files": [
|
||||
"Chatterbox-GGUF/chatterbox-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Chatterbox-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "chatterbox_f16",
|
||||
"display_name": "Chatterbox F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Chatterbox-GGUF",
|
||||
"files": [
|
||||
"Chatterbox-GGUF/chatterbox-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Chatterbox-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "chatterbox_safetensors",
|
||||
"display_name": "Chatterbox Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "chatterbox",
|
||||
"files": [
|
||||
"ve.safetensors",
|
||||
"t3_cfg.safetensors",
|
||||
"s3gen.safetensors",
|
||||
"tokenizer.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "ResembleAI/chatterbox"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"english_tokenizer": "model:tokenizer.json",
|
||||
"multilingual_tokenizer": "model:grapheme_mtl_merged_expanded_v1.json",
|
||||
"cangjie_mapping": "model:Cangjie5_TC.json",
|
||||
"builtin_conditionals": "model:conds.pt"
|
||||
},
|
||||
"tensors": {
|
||||
"voice_encoder_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "voice_encoder"
|
||||
},
|
||||
"s3gen_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "s3gen"
|
||||
},
|
||||
"t3_english_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "t3_english"
|
||||
},
|
||||
"t3_multilingual_v2_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "t3_multilingual_v2"
|
||||
},
|
||||
"t3_multilingual_v3_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "t3_multilingual_v3"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"english_tokenizer": "model:tokenizer.json",
|
||||
"multilingual_tokenizer": "model:grapheme_mtl_merged_expanded_v1.json",
|
||||
"cangjie_mapping": "model:Cangjie5_TC.json",
|
||||
"builtin_conditionals": "model:conds.pt"
|
||||
},
|
||||
"tensors": {
|
||||
"voice_encoder_weights": "model:ve.safetensors",
|
||||
"s3gen_weights": "model:s3gen.safetensors",
|
||||
"t3_english_weights": "model:t3_cfg.safetensors",
|
||||
"t3_multilingual_v2_weights": "model:t3_mtl23ls_v2.safetensors",
|
||||
"t3_multilingual_v3_weights": "model:t3_mtl23ls_v3.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
{
|
||||
"family": "citrinet_asr",
|
||||
"display_name": "Citrinet ASR",
|
||||
"description": "NVIDIA CitriNet-family end-to-end English ASR model using a convolutional CTC architecture optimized for transcribing speech segments to text.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en"
|
||||
],
|
||||
"capabilities": {},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "citrinet_asr_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "citrinet_asr_q8_0",
|
||||
"display_name": "Citrinet ASR Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Citrinet-ASR-GGUF",
|
||||
"files": [
|
||||
"Citrinet-ASR-GGUF/citrinet-asr-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Citrinet-ASR-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:citrinet_256_config.json",
|
||||
"tokenizer": "model:citrinet_256_tokenizer.model"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:citrinet_256_config.json",
|
||||
"tokenizer": "model:citrinet_256_tokenizer.model"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:citrinet_256.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,330 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "confucius4_tts",
|
||||
"display_name": "Confucius4-TTS",
|
||||
"description": "Confucius4-TTS is a multilingual voice-cloning TTS model packaged for audio.cpp with offline and streaming generation. It uses reference speech, language-aware text normalization, T2S semantic generation, S2A flow matching, style encoding, semantic audio features, and BigVGAN vocoding.",
|
||||
"category": "tts",
|
||||
"status": "experimental",
|
||||
"tasks": [
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"zh",
|
||||
"en",
|
||||
"ja",
|
||||
"ko",
|
||||
"de",
|
||||
"fr",
|
||||
"es",
|
||||
"id",
|
||||
"it",
|
||||
"th",
|
||||
"pt",
|
||||
"ru",
|
||||
"ms",
|
||||
"vi"
|
||||
],
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference",
|
||||
"long_form"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "language",
|
||||
"type": "string",
|
||||
"description": "Target synthesis language code used by the text frontend; default zh when no request transcript language or style language is provided.",
|
||||
"required": false,
|
||||
"default": "zh"
|
||||
},
|
||||
{
|
||||
"name": "temperature",
|
||||
"type": "float",
|
||||
"description": "T2S sampling temperature; must be positive, default 0.8.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.8
|
||||
},
|
||||
{
|
||||
"name": "top_p",
|
||||
"type": "float",
|
||||
"description": "T2S nucleus sampling probability; must be in (0, 1], default 0.8.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"max": 1.0,
|
||||
"default": 0.8
|
||||
},
|
||||
{
|
||||
"name": "top_k",
|
||||
"type": "int",
|
||||
"description": "T2S top-k sampling limit; must be positive, default 30.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 30
|
||||
},
|
||||
{
|
||||
"name": "num_beams",
|
||||
"type": "int",
|
||||
"description": "T2S beam count; default 3. Set 1 for single-beam sampling.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 3
|
||||
},
|
||||
{
|
||||
"name": "repetition_penalty",
|
||||
"type": "float",
|
||||
"description": "T2S repetition penalty; must be positive, default 10.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 10.0
|
||||
},
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum T2S semantic sequence length including prompt tokens; default 1520.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1520
|
||||
},
|
||||
{
|
||||
"name": "num_inference_steps",
|
||||
"type": "int",
|
||||
"description": "S2A flow-matching step count; default 25.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 25
|
||||
},
|
||||
{
|
||||
"name": "guidance_scale",
|
||||
"type": "float",
|
||||
"description": "S2A classifier-free guidance scale; default 0.7.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.7
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_size",
|
||||
"type": "int",
|
||||
"description": "Maximum text tokens per generated segment; default 80.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 80
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Framework text chunking mode; default uses the standard word-budget chunker.",
|
||||
"values": [
|
||||
"default",
|
||||
"tag_aware",
|
||||
"japanese",
|
||||
"endline"
|
||||
],
|
||||
"required": false,
|
||||
"default": "default"
|
||||
},
|
||||
{
|
||||
"name": "cross_fade_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Cross-fade duration between generated segments in seconds; default 0.3.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.3
|
||||
},
|
||||
{
|
||||
"name": "edge_fade_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Fade duration applied at segment edges in seconds; default 0.1.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.1
|
||||
},
|
||||
{
|
||||
"name": "edge_pad_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Silence padding applied at segment edges in seconds; default 0.1.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.1
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Seed for T2S sampling and S2A noise initialization; default 1234.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 1234
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Reusable ggml graph arena size in MiB for Confucius stages; default 512.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Weight loading context size in MiB; default 1024.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1024
|
||||
},
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Matmul weight storage type; default native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "conv_weight_type",
|
||||
"type": "enum",
|
||||
"description": "Convolution weight storage type; default native.",
|
||||
"preset": "weight_type_conv",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "reference_cache_slots",
|
||||
"type": "int",
|
||||
"description": "Prepared reference-audio cache slots; default 1, set 0 to disable caching.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 1
|
||||
},
|
||||
{
|
||||
"name": "mem_saver",
|
||||
"type": "bool",
|
||||
"description": "Release staged graphs after request phases; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "confucius4_tts_orig",
|
||||
"display_name": "Confucius4-TTS Original-Dtype GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "Confucius4-TTS-GGUF",
|
||||
"files": [
|
||||
"Confucius4-TTS-GGUF/confucius4-tts-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "Confucius4-TTS-GGUF"
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "confucius4_tts_orig",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:inference_config.yaml",
|
||||
"tokenizer_model": "model:tokenizer.model",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json",
|
||||
"w2v_preprocessor_config": "model:w2v_preprocessor_config.json",
|
||||
"bigvgan_config": "model:bigvgan_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"t2s": {
|
||||
"source": "weights:",
|
||||
"prefix": "t2s"
|
||||
},
|
||||
"s2a": {
|
||||
"source": "weights:",
|
||||
"prefix": "s2a"
|
||||
},
|
||||
"semantic_encoder": {
|
||||
"source": "weights:",
|
||||
"prefix": "semantic_encoder"
|
||||
},
|
||||
"semantic_encoder_shaw": {
|
||||
"source": "weights:",
|
||||
"prefix": "semantic_encoder_shaw"
|
||||
},
|
||||
"semantic_stats": {
|
||||
"source": "weights:",
|
||||
"prefix": "semantic_stats"
|
||||
},
|
||||
"style_encoder": {
|
||||
"source": "weights:",
|
||||
"prefix": "style_encoder"
|
||||
},
|
||||
"vocoder": {
|
||||
"source": "weights:",
|
||||
"prefix": "vocoder"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:inference_config.yaml",
|
||||
"tokenizer_model": "model:tokenizer.model",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json",
|
||||
"w2v_preprocessor_config": "model:w2v_preprocessor_config.json",
|
||||
"bigvgan_config": "model:bigvgan_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"t2s": "model:t2s.safetensors",
|
||||
"s2a": "model:s2a.safetensors",
|
||||
"semantic_encoder": "model:semantic_encoder.safetensors",
|
||||
"semantic_encoder_shaw": "model:semantic_encoder_shaw.safetensors",
|
||||
"semantic_stats": "model:semantic_stats.safetensors",
|
||||
"style_encoder": "model:style_encoder.safetensors",
|
||||
"vocoder": "model:vocoder.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,229 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "dramabox",
|
||||
"display_name": "DramaBox",
|
||||
"description": "DramaBox is an English expressive TTS and voice-cloning model packaged for audio.cpp as a standalone GGUF bundle. It combines Gemma text conditioning, diffusion sampling, reference-audio conditioning, long-form chunking, and 48 kHz stereo output.",
|
||||
"category": "tts",
|
||||
"status": "experimental",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en"
|
||||
],
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"capabilities": {
|
||||
"tts": [
|
||||
"speaker_reference",
|
||||
"style_control",
|
||||
"long_form"
|
||||
],
|
||||
"clone": [
|
||||
"speaker_reference",
|
||||
"style_control",
|
||||
"long_form"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "target_voice",
|
||||
"type": "audio_path",
|
||||
"description": "Reference voice WAV path for voice cloning. If omitted, DramaBox generates from text without reference-audio conditioning.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "negative_prompt",
|
||||
"type": "string",
|
||||
"description": "Negative text conditioning used when guidance_scale enables classifier-free guidance; omitted uses the built-in quality prompt.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "duration_sec",
|
||||
"type": "float",
|
||||
"description": "Explicit target duration in seconds; default 0 uses the prompt-duration estimator.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.0
|
||||
},
|
||||
{
|
||||
"name": "num_inference_steps",
|
||||
"type": "int",
|
||||
"description": "Diffusion sampling step count; default comes from config.json, 30 in the current package.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 30
|
||||
},
|
||||
{
|
||||
"name": "guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Classifier-free guidance scale; default comes from config.json, 2.5 in the current package. Values greater than 1 enable CFG.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 2.5
|
||||
},
|
||||
{
|
||||
"name": "spatio_temporal_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Spatio-temporal guidance scale; default comes from config.json, 1.5 in the current package. Values greater than 0 enable STG.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.5
|
||||
},
|
||||
{
|
||||
"name": "duration_scale",
|
||||
"type": "float",
|
||||
"description": "Multiplier applied to the estimated prompt duration when duration_sec is 0; default comes from config.json, 1.1 in the current package.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.1
|
||||
},
|
||||
{
|
||||
"name": "reference_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Reference voice crop/repeat duration in seconds; default comes from config.json, 10.0 in the current package.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 10.0
|
||||
},
|
||||
{
|
||||
"name": "guidance_rescale",
|
||||
"type": "string",
|
||||
"description": "Guidance rescale value. The default auto mode derives a rescale value from guidance_scale; a numeric string requests an explicit value.",
|
||||
"required": false,
|
||||
"default": "auto"
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_threshold_sec",
|
||||
"type": "float",
|
||||
"description": "Estimated duration threshold that switches a request to long-form chunking; default 45.0 seconds.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 45.0
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Target estimated duration for each long-form chunk; default 37.0 seconds.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 37.0
|
||||
},
|
||||
{
|
||||
"name": "cross_fade_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Equal-power cross-fade between long-form chunks in seconds; default 0.05.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.05
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Torch-compatible CUDA noise seed for diffusion sampling; default 42.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 42
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "perf_mode",
|
||||
"type": "enum",
|
||||
"description": "Attention implementation mode. Default off keeps the exact reference-query attention path; flash_attention enables the optimized path.",
|
||||
"values": [
|
||||
"off",
|
||||
"flash_attention"
|
||||
],
|
||||
"required": false,
|
||||
"default": "off"
|
||||
},
|
||||
{
|
||||
"name": "mem_saver",
|
||||
"type": "bool",
|
||||
"description": "Release staged runtime graphs and weights immediately after each request phase to reduce peak and resident VRAM; default false keeps components cached for later reuse.",
|
||||
"required": false,
|
||||
"default": false
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "dramabox_q8_0",
|
||||
"display_name": "DramaBox Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "DramaBox-GGUF",
|
||||
"files": [
|
||||
"DramaBox-GGUF/dramabox-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "DramaBox-GGUF"
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "dramabox_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"audio_components_config": "model:audio_components_config.json",
|
||||
"gemma_config": "model:gemma-3-12b-it-bnb-4bit/config.json",
|
||||
"gemma_tokenizer_model": "model:gemma-3-12b-it-bnb-4bit/tokenizer.model",
|
||||
"gemma_tokenizer_json": "model:gemma-3-12b-it-bnb-4bit/tokenizer.json",
|
||||
"gemma_tokenizer_config": "model:gemma-3-12b-it-bnb-4bit/tokenizer_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"dit_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "dit"
|
||||
},
|
||||
"audio_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "audio"
|
||||
},
|
||||
"gemma_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "gemma"
|
||||
},
|
||||
"silence_latent": {
|
||||
"source": "weights:",
|
||||
"prefix": "silence"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
{
|
||||
"family": "fish_audio",
|
||||
"display_name": "Fish Audio S2 Pro",
|
||||
"description": "Fish Audio S2 Pro text-to-speech model for expressive speech across 80+ languages, with automatic language handling, inline prosody/emotion controls, and rapid voice cloning from short reference samples.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"80+ languages"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "fish_audio_s2_pro_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "fish_audio_s2_pro_q8_0",
|
||||
"display_name": "Fish Audio S2 Pro Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Fish-Audio-S2-Pro-GGUF",
|
||||
"files": [
|
||||
"Fish-Audio-S2-Pro-GGUF/fish-audio-s2-pro-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Fish-Audio-S2-Pro-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "fish_audio_s2_pro_bf16",
|
||||
"display_name": "Fish Audio S2 Pro BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "Fish-Audio-S2-Pro-GGUF",
|
||||
"files": [
|
||||
"Fish-Audio-S2-Pro-GGUF/fish-audio-s2-pro-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "Fish-Audio-S2-Pro-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"codec_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "codec_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model_audio_cpp.safetensors.index.json",
|
||||
"codec_weights": "model:codec.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,201 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "fun_asr_nano",
|
||||
"display_name": "Fun-ASR-Nano",
|
||||
"description": "Offline multilingual speech recognition with the FunAudioLLM Fun-ASR-Nano-2512 model.",
|
||||
"category": "asr",
|
||||
"status": "wip",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"auto",
|
||||
"zh",
|
||||
"en",
|
||||
"ja"
|
||||
],
|
||||
"capabilities": {},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "language",
|
||||
"type": "string",
|
||||
"description": "Recognition language, or auto to let the model infer it.",
|
||||
"required": false,
|
||||
"default": "auto"
|
||||
},
|
||||
{
|
||||
"name": "enable_itn",
|
||||
"type": "bool",
|
||||
"description": "Enable inverse text normalization in the transcription prompt.",
|
||||
"required": false,
|
||||
"default": true
|
||||
},
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum number of generated transcript tokens.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Audio chunking mode: auto, fixed, or none.",
|
||||
"values": [
|
||||
"auto",
|
||||
"fixed",
|
||||
"none"
|
||||
],
|
||||
"required": false,
|
||||
"default": "auto"
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_seconds",
|
||||
"type": "float",
|
||||
"description": "Fixed chunk duration in seconds.",
|
||||
"required": false,
|
||||
"min": 0.001,
|
||||
"default": 30
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Shared model weight storage type.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"server",
|
||||
"cuda",
|
||||
"metal",
|
||||
"cpu"
|
||||
]
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "fun_asr_nano_2512_q8_0",
|
||||
"display_name": "Fun-ASR-Nano-2512 Q8_0 GGUF",
|
||||
"description": "Standalone audio.cpp GGUF built from the pinned official checkpoint; governed by the FunASR Model Open Source License Agreement v1.1.",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Fun-ASR-Nano-2512-GGUF",
|
||||
"files": [
|
||||
"fun-asr-nano-2512-q8_0.gguf"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "FunAudioLLM/Fun-ASR-Nano-2512-GGUF",
|
||||
"revision": "ce72677f84900f0dc57f498ace253bfb3c9155b6",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "fun_asr_nano_2512_f16",
|
||||
"display_name": "Fun-ASR-Nano-2512 F16 GGUF",
|
||||
"description": "Standalone audio.cpp GGUF built from the pinned official checkpoint; governed by the FunASR Model Open Source License Agreement v1.1.",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Fun-ASR-Nano-2512-GGUF",
|
||||
"files": [
|
||||
"fun-asr-nano-2512-f16.gguf"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "FunAudioLLM/Fun-ASR-Nano-2512-GGUF",
|
||||
"revision": "ce72677f84900f0dc57f498ace253bfb3c9155b6",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "fun_asr_nano_2512_safetensors",
|
||||
"display_name": "Fun-ASR-Nano-2512 HF Safetensors",
|
||||
"description": "Official checkpoint governed by the FunASR Model Open Source License Agreement v1.1.",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "Fun-ASR-Nano-2512-hf",
|
||||
"files": [
|
||||
"chat_template.jinja",
|
||||
"config.json",
|
||||
"generation_config.json",
|
||||
"model.safetensors",
|
||||
"processor_config.json",
|
||||
"tokenizer.json",
|
||||
"tokenizer_config.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "FunAudioLLM/Fun-ASR-Nano-2512-hf",
|
||||
"revision": "854d88f94205cd17d2afdb24332130d86fbe654a",
|
||||
"gated": false
|
||||
}
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "fun_asr_nano_2512_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
],
|
||||
"summary": "Offline Fun-ASR-Nano transcription from official safetensors or audio.cpp GGUF."
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"chat_template_jinja": "model:chat_template.jinja",
|
||||
"tokenizer_config": "model:tokenizer_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"chat_template_jinja": "model:chat_template.jinja",
|
||||
"tokenizer_config": "model:tokenizer_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,260 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "glm_tts",
|
||||
"display_name": "GLM-TTS",
|
||||
"description": "Community Chinese-English zero-shot speech synthesis and voice cloning with native Llama, Whisper-VQ, Flow/DiT, CAMPPlus, and HiFT execution.",
|
||||
"category": "tts",
|
||||
"status": "community",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"zh",
|
||||
"en"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "reference_text",
|
||||
"type": "string",
|
||||
"description": "Transcript matching the reference voice audio; required by GLM-TTS zero-shot synthesis.",
|
||||
"required": true
|
||||
},
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum generated speech tokens; otherwise the official 2x-to-20x text-token bounds are used.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 0
|
||||
},
|
||||
{
|
||||
"name": "temperature",
|
||||
"type": "float",
|
||||
"description": "Speech-token temperature; official default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "top_k",
|
||||
"type": "int",
|
||||
"description": "Speech-token top-k; official default 25.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 25
|
||||
},
|
||||
{
|
||||
"name": "top_p",
|
||||
"type": "float",
|
||||
"description": "Speech-token nucleus threshold; official default 0.8.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"max": 1.0,
|
||||
"default": 0.8
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Speech-token, Flow-noise, and HiFT seed.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 0
|
||||
},
|
||||
{
|
||||
"name": "num_inference_steps",
|
||||
"type": "int",
|
||||
"description": "Flow Euler steps; defaults to model config, usually official default 10.",
|
||||
"required": false,
|
||||
"min": 1
|
||||
},
|
||||
{
|
||||
"name": "flow_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Flow classifier-free guidance rate; defaults to model config, usually official default 0.7.",
|
||||
"required": false,
|
||||
"min": 0.0
|
||||
},
|
||||
{
|
||||
"name": "flow_noise_path",
|
||||
"type": "path",
|
||||
"description": "Optional raw float32 initial Flow noise for parity tests.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "hift_source_random_path",
|
||||
"type": "path",
|
||||
"description": "Optional raw float32 HiFT phase-uniform and Gaussian values for parity tests.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "hift_prior_noise_count",
|
||||
"type": "int",
|
||||
"description": "Torch RNG value offset used before HiFT source generation; default 0.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 0
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Requested component weight storage type; default native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "mem_saver",
|
||||
"type": "bool",
|
||||
"description": "Release reference-only encoders after caching the voice while keeping the generation path warm; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "aggressive_mem_saver",
|
||||
"type": "bool",
|
||||
"description": "Also release Llama, Flow, and HiFT after every stage. Minimizes VRAM but reloads the generation path on every request; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "reference_cache_slots",
|
||||
"type": "int",
|
||||
"description": "Prepared reference-audio cache slots; default 1. Use 0 to disable.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 1
|
||||
},
|
||||
{
|
||||
"name": "llama_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Llama weight metadata context in MiB; default 8192.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 8192
|
||||
},
|
||||
{
|
||||
"name": "constant_context_mb",
|
||||
"type": "int",
|
||||
"description": "Llama constant tensor context in MiB; default 256.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 256
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"llama_config": "model:llm/config.json",
|
||||
"llama_generation_config": "model:llm/generation_config.json",
|
||||
"speech_tokenizer_config": "model:speech_tokenizer/config.json",
|
||||
"speech_tokenizer_preprocessor": "model:speech_tokenizer/preprocessor_config.json",
|
||||
"tokenizer_config": "model:vq32k-phoneme-tokenizer/tokenizer_config.json",
|
||||
"tokenizer_vocab": "model:vq32k-phoneme-tokenizer/tokenizer_vocab.json",
|
||||
"tokenizer_merges": "model:vq32k-phoneme-tokenizer/tokenizer_merges.txt",
|
||||
"flow_config": "model:flow/config.yaml",
|
||||
"audio_cpp_config": "model:audio_cpp_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"llama_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "llama_weights"
|
||||
},
|
||||
"speech_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "speech_tokenizer_weights"
|
||||
},
|
||||
"flow_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "flow_weights"
|
||||
},
|
||||
"hift_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "hift_weights"
|
||||
},
|
||||
"campplus_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "campplus_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"llama_config": "model:llm/config.json",
|
||||
"llama_generation_config": "model:llm/generation_config.json",
|
||||
"speech_tokenizer_config": "model:speech_tokenizer/config.json",
|
||||
"speech_tokenizer_preprocessor": "model:speech_tokenizer/preprocessor_config.json",
|
||||
"tokenizer_config": "model:vq32k-phoneme-tokenizer/tokenizer_config.json",
|
||||
"tokenizer_vocab": "model:vq32k-phoneme-tokenizer/tokenizer_vocab.json",
|
||||
"tokenizer_merges": "model:vq32k-phoneme-tokenizer/tokenizer_merges.txt",
|
||||
"flow_config": "model:flow/config.yaml",
|
||||
"audio_cpp_config": "model:audio_cpp_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"llama_weights": "model:llm/model.safetensors.index.json",
|
||||
"speech_tokenizer_weights": "model:speech_tokenizer/model.safetensors",
|
||||
"flow_weights": "model:flow/model.safetensors",
|
||||
"hift_weights": "model:hift/model.safetensors",
|
||||
"campplus_weights": "model:frontend/campplus.safetensors"
|
||||
}
|
||||
}
|
||||
],
|
||||
"packages": [
|
||||
{
|
||||
"id": "glm_tts_q8_0",
|
||||
"display_name": "GLM-TTS mixed Q8_0/F16 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "GLM-TTS-Q8",
|
||||
"files": [
|
||||
"Text to audio (TTS)/GLM-TTS_Q8.gguf"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "mirek190/audio.cpp"
|
||||
}
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "glm_tts_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/community_models/glm_tts.md",
|
||||
"docs/reports/glm_tts_validation.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "higgs_audio_stt",
|
||||
"display_name": "Higgs Audio v3 STT",
|
||||
"description": "Boson AI English speech-to-text model combining a Whisper Large v3 speech encoder with a Qwen decoder for robust ASR.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"en"
|
||||
],
|
||||
"capabilities": {},
|
||||
"dependencies": [],
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "language",
|
||||
"type": "string",
|
||||
"description": "Transcript language code metadata; English is used when omitted.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum generated transcript tokens; default 1024.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1024
|
||||
},
|
||||
{
|
||||
"name": "enable_thinking",
|
||||
"type": "bool",
|
||||
"description": "Enable the model thinking prompt; default true.",
|
||||
"required": false,
|
||||
"default": true
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Audio chunking mode; default auto uses fixed chunks.",
|
||||
"values": [
|
||||
"auto",
|
||||
"fixed",
|
||||
"none"
|
||||
],
|
||||
"required": false,
|
||||
"default": "auto"
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Fixed audio chunk duration in seconds; must be positive when set; default 4.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 4.0
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Shared text decoder weight storage type; default native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "audio_encoder_weight_type",
|
||||
"type": "enum",
|
||||
"description": "Audio encoder convolution weight storage type; default native.",
|
||||
"preset": "weight_type_conv",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "text_decoder_weight_type",
|
||||
"type": "enum",
|
||||
"description": "Text decoder matmul weight storage type; defaults to weight_type when set, otherwise native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "audio_encoder_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Audio encoder graph arena size in MiB; default 512.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "text_decoder_prefill_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Text decoder prefill graph arena size in MiB; default 512.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "text_decoder_decode_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Text decoder cached-step graph arena size in MiB; default 256.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 256
|
||||
},
|
||||
{
|
||||
"name": "text_decoder_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Text decoder weight context arena size in MiB; default 4096.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 4096
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "higgs_audio_stt_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "higgs_audio_stt_q8_0",
|
||||
"display_name": "Higgs Audio v3 STT Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Higgs-Audio-v3-STT-GGUF",
|
||||
"files": [
|
||||
"Higgs-Audio-v3-STT-GGUF/higgs-audio-v3-stt-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Higgs-Audio-v3-STT-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "higgs_audio_stt_f16",
|
||||
"display_name": "Higgs Audio v3 STT F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Higgs-Audio-v3-STT-GGUF",
|
||||
"files": [
|
||||
"Higgs-Audio-v3-STT-GGUF/higgs-audio-v3-stt-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Higgs-Audio-v3-STT-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"preprocessor_config": "model:preprocessor_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"whisper": "../whisper-large-v3"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"preprocessor_config": "whisper:preprocessor_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors.index.json"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
{
|
||||
"family": "higgs_audio_tts",
|
||||
"display_name": "Higgs Audio v3 TTS",
|
||||
"description": "Boson AI conversational TTS model for expressive speech across 100+ languages, zero-shot voice cloning, and inline control over emotion, style, prosody, pauses, and sound effects.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"100+ languages"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "higgs_audio_tts_4b_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "higgs_audio_tts_4b_q8_0",
|
||||
"display_name": "Higgs Audio v3 TTS 4B Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Higgs-Audio-v3-TTS-4B-GGUF",
|
||||
"files": [
|
||||
"Higgs-Audio-v3-TTS-4B-GGUF/higgs-audio-v3-tts-4b-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Higgs-Audio-v3-TTS-4B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "higgs_audio_tts_4b_bf16",
|
||||
"display_name": "Higgs Audio v3 TTS 4B BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "Higgs-Audio-v3-TTS-4B-GGUF",
|
||||
"files": [
|
||||
"Higgs-Audio-v3-TTS-4B-GGUF/higgs-audio-v3-tts-4b-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "Higgs-Audio-v3-TTS-4B-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"chat_template": "model:chat_template.jinja"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"chat_template": "model:chat_template.jinja"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors.index.json"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,265 @@
|
||||
{
|
||||
"family": "hviske_asr",
|
||||
"schema_version": 1,
|
||||
"display_name": "Hviske ASR",
|
||||
"description": "Danish-optimized Conformer encoder-decoder ASR model fine-tuned from the Hviske v5 family, with selectable Cohere ASR language prompts.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"ar",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"en",
|
||||
"es",
|
||||
"fr",
|
||||
"it",
|
||||
"ja",
|
||||
"ko",
|
||||
"nl",
|
||||
"pl",
|
||||
"pt",
|
||||
"vi",
|
||||
"zh"
|
||||
],
|
||||
"capabilities": {},
|
||||
"dependencies": [],
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "language",
|
||||
"type": "string",
|
||||
"description": "ASR language code; default da.",
|
||||
"required": false,
|
||||
"default": "da"
|
||||
},
|
||||
{
|
||||
"name": "punctuation",
|
||||
"type": "bool",
|
||||
"description": "Enable or disable punctuation tokens in the decoder prompt; default true.",
|
||||
"required": false,
|
||||
"default": true
|
||||
},
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum generated transcript tokens; defaults to the model config.",
|
||||
"required": false,
|
||||
"min": 1
|
||||
},
|
||||
{
|
||||
"name": "num_beams",
|
||||
"type": "int",
|
||||
"description": "Beam-search beam count; default 1 uses greedy or sampling decode.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1
|
||||
},
|
||||
{
|
||||
"name": "length_penalty",
|
||||
"type": "float",
|
||||
"description": "Beam-search length penalty; must be positive when set; default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.000001,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "do_sample",
|
||||
"type": "bool",
|
||||
"description": "Enable sampling instead of greedy decode when num_beams is 1; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "temperature",
|
||||
"type": "float",
|
||||
"description": "Decoder sampling temperature; must be positive when set; default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.000001,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "top_k",
|
||||
"type": "int",
|
||||
"description": "Top-k sampling limit; default 50, 0 disables top-k.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 50
|
||||
},
|
||||
{
|
||||
"name": "top_p",
|
||||
"type": "float",
|
||||
"description": "Nucleus sampling limit in (0, 1]; default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.000001,
|
||||
"max": 1.0,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Decoder sampling seed; random if omitted.",
|
||||
"required": false,
|
||||
"min": 0
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Audio chunking mode; default auto uses quiet-energy splitting only when audio exceeds the model clip window.",
|
||||
"values": [
|
||||
"auto",
|
||||
"fixed",
|
||||
"quiet_energy",
|
||||
"none"
|
||||
],
|
||||
"required": false,
|
||||
"default": "auto"
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Maximum audio chunk duration in seconds; defaults to the model clip window.",
|
||||
"required": false,
|
||||
"min": 0.000001
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Matmul weight storage type; default native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "conv_weight_type",
|
||||
"type": "enum",
|
||||
"description": "Convolution weight storage type; defaults to weight_type when set, otherwise native.",
|
||||
"preset": "weight_type_conv",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Weight descriptor context size in MiB; default 32.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 32
|
||||
},
|
||||
{
|
||||
"name": "encoder_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Encoder graph arena size in MiB; default 512.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "decoder_prefill_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Decoder prefill graph arena size in MiB; default 512.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "decoder_decode_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Decoder cached-step graph arena size in MiB; default 512.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 512
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "hviske_asr_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "hviske_asr_q8_0",
|
||||
"display_name": "Hviske v5.3 Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Hviske-v5.3-GGUF",
|
||||
"files": [
|
||||
"Audio to text (ASR)/hviske-v5.3_Q8.gguf"
|
||||
],
|
||||
"strip_prefix": "Audio to text (ASR)",
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "mirek190/audio.cpp"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "hviske_asr_safetensors",
|
||||
"display_name": "Hviske v5.3 Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "hviske-v5.3",
|
||||
"files": [
|
||||
"config.json",
|
||||
"generation_config.json",
|
||||
"model.safetensors",
|
||||
"tokenizer.model"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "syvai/hviske-v5.3"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer": "model:tokenizer.model"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer": "model:tokenizer.model"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,199 @@
|
||||
{
|
||||
"family": "index_tts2",
|
||||
"display_name": "IndexTTS2",
|
||||
"description": "Zero-shot TTS system for Chinese and English speech synthesis with voice cloning, emotion-speaker decoupling, text or audio emotion control, and explicit duration control.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"zh",
|
||||
"en"
|
||||
],
|
||||
"capabilities": {
|
||||
"tts": [
|
||||
"emotion_control"
|
||||
],
|
||||
"clone": [
|
||||
"speaker_reference",
|
||||
"emotion_control"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "index_tts2_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "index_tts2_q8_0",
|
||||
"display_name": "IndexTTS2 Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "IndexTTS2-GGUF",
|
||||
"files": [
|
||||
"IndexTTS2-GGUF/index-tts2-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "IndexTTS2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "index_tts2_f16",
|
||||
"display_name": "IndexTTS2 F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "IndexTTS2-GGUF",
|
||||
"files": [
|
||||
"IndexTTS2-GGUF/index-tts2-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "IndexTTS2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "index_tts2_orig",
|
||||
"display_name": "IndexTTS2 Original-Dtype GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "IndexTTS2-GGUF",
|
||||
"files": [
|
||||
"IndexTTS2-GGUF/index-tts2-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "IndexTTS2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "index_tts2_safetensors",
|
||||
"display_name": "IndexTTS2 Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "IndexTTS-2",
|
||||
"files": [
|
||||
"config.yaml",
|
||||
"bpe.model",
|
||||
"gpt.safetensors"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "mlx-community/index-tts2-mlx"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.yaml",
|
||||
"bpe": "model:bpe.model",
|
||||
"wav2vec2bert_config": "model:w2v-bert-2.0/config.json",
|
||||
"wav2vec2bert_preprocessor_config": "model:w2v-bert-2.0/preprocessor_config.json",
|
||||
"bigvgan_config": "model:bigvgan/config.json",
|
||||
"qwen_emotion_config": "model:qwen0.6bemo4-merge/config.json",
|
||||
"qwen_emotion_generation_config": "model:qwen0.6bemo4-merge/generation_config.json",
|
||||
"qwen_emotion_tokenizer": "model:qwen0.6bemo4-merge/tokenizer.json",
|
||||
"qwen_emotion_tokenizer_config": "model:qwen0.6bemo4-merge/tokenizer_config.json",
|
||||
"qwen_emotion_vocab": "model:qwen0.6bemo4-merge/vocab.json",
|
||||
"qwen_emotion_merges": "model:qwen0.6bemo4-merge/merges.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"gpt": {
|
||||
"source": "weights:",
|
||||
"prefix": "gpt"
|
||||
},
|
||||
"s2mel": {
|
||||
"source": "weights:",
|
||||
"prefix": "s2mel"
|
||||
},
|
||||
"speaker_matrix": {
|
||||
"source": "weights:",
|
||||
"prefix": "speaker_matrix"
|
||||
},
|
||||
"emotion_matrix": {
|
||||
"source": "weights:",
|
||||
"prefix": "emotion_matrix"
|
||||
},
|
||||
"wav2vec2bert_stats": {
|
||||
"source": "weights:",
|
||||
"prefix": "wav2vec2bert_stats"
|
||||
},
|
||||
"wav2vec2bert": {
|
||||
"source": "weights:",
|
||||
"prefix": "wav2vec2bert"
|
||||
},
|
||||
"semantic_codec": {
|
||||
"source": "weights:",
|
||||
"prefix": "semantic_codec"
|
||||
},
|
||||
"campplus": {
|
||||
"source": "weights:",
|
||||
"prefix": "campplus"
|
||||
},
|
||||
"bigvgan": {
|
||||
"source": "weights:",
|
||||
"prefix": "bigvgan"
|
||||
},
|
||||
"qwen_emotion": {
|
||||
"source": "weights:",
|
||||
"prefix": "qwen_emotion"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.yaml",
|
||||
"bpe": "model:bpe.model",
|
||||
"wav2vec2bert_config": "model:w2v-bert-2.0/config.json",
|
||||
"wav2vec2bert_preprocessor_config": "model:w2v-bert-2.0/preprocessor_config.json",
|
||||
"bigvgan_config": "model:bigvgan/config.json",
|
||||
"qwen_emotion_config": "model:qwen0.6bemo4-merge/config.json",
|
||||
"qwen_emotion_generation_config": "model:qwen0.6bemo4-merge/generation_config.json",
|
||||
"qwen_emotion_tokenizer": "model:qwen0.6bemo4-merge/tokenizer.json",
|
||||
"qwen_emotion_tokenizer_config": "model:qwen0.6bemo4-merge/tokenizer_config.json",
|
||||
"qwen_emotion_vocab": "model:qwen0.6bemo4-merge/vocab.json",
|
||||
"qwen_emotion_merges": "model:qwen0.6bemo4-merge/merges.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"gpt": "model:gpt.safetensors",
|
||||
"s2mel": "model:s2mel.safetensors",
|
||||
"speaker_matrix": "model:feat1.safetensors",
|
||||
"emotion_matrix": "model:feat2.safetensors",
|
||||
"wav2vec2bert_stats": "model:wav2vec2bert_stats.safetensors",
|
||||
"wav2vec2bert": "model:w2v-bert-2.0/model.safetensors",
|
||||
"semantic_codec": "model:semantic_codec_model.safetensors",
|
||||
"campplus": "model:campplus.safetensors",
|
||||
"bigvgan": "model:bigvgan/model.safetensors",
|
||||
"qwen_emotion": "model:qwen0.6bemo4-merge/model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "inflect_v2",
|
||||
"display_name": "Inflect Micro v2",
|
||||
"description": "Compact English VITS text-to-speech models with a native GGML inference path and an external eSpeak-ng phonemizer.",
|
||||
"category": "tts",
|
||||
"status": "community",
|
||||
"tasks": [
|
||||
"tts"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en"
|
||||
],
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"capabilities": {
|
||||
"tts": [
|
||||
"long_form"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "speaking_rate",
|
||||
"type": "float",
|
||||
"description": "Speech speed multiplier mapped directly to Inflect speed; default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.5,
|
||||
"max": 2.0,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "variation",
|
||||
"type": "float",
|
||||
"description": "Latent Gaussian variation; default 0.667.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"max": 1.0,
|
||||
"default": 0.667
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Non-negative latent noise seed; default 0.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 0
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Long-form text chunking mode.",
|
||||
"values": [
|
||||
"word_budget"
|
||||
],
|
||||
"required": false,
|
||||
"default": "word_budget"
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_size",
|
||||
"type": "int",
|
||||
"description": "Maximum Unicode codepoints per long-form text chunk; default 280.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 280
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "espeak_library_path",
|
||||
"type": "path",
|
||||
"description": "Optional explicit path to the eSpeak-ng shared library.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "espeak_data_path",
|
||||
"type": "path",
|
||||
"description": "Optional explicit path to the directory containing espeak-ng-data.",
|
||||
"required": false
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "inflect_micro_v2_orig",
|
||||
"display_name": "Inflect Micro v2 Original-Dtype GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "Inflect-Micro-v2-GGUF",
|
||||
"files": [
|
||||
"Inflect-Micro-v2-GGUF/inflect-micro-v2-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "Inflect-Micro-v2-GGUF"
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "inflect_micro_v2_orig",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/community_models/inflect_v2.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,406 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "irodori_tts",
|
||||
"display_name": "Irodori-TTS",
|
||||
"description": "Japanese TTS model based on RF-DiT continuous audio latents, supporting zero-shot voice cloning, automatic duration prediction, multimodal voice design, and emoji-style control.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone",
|
||||
"design"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"ja"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"design": [
|
||||
"voice_design"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "language",
|
||||
"type": "string",
|
||||
"description": "Text language code; Irodori-TTS supports Japanese only.",
|
||||
"required": false,
|
||||
"default": "ja"
|
||||
},
|
||||
{
|
||||
"name": "caption",
|
||||
"type": "string",
|
||||
"description": "VoiceDesign caption describing target voice identity, style, or emotion; supported only by caption-conditioned checkpoints.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "no_ref",
|
||||
"type": "bool",
|
||||
"description": "Use no-reference generation; default true unless a speaker reference is provided.",
|
||||
"required": false,
|
||||
"default": true
|
||||
},
|
||||
{
|
||||
"name": "num_inference_steps",
|
||||
"type": "int",
|
||||
"description": "RF diffusion steps; default 40.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 40
|
||||
},
|
||||
{
|
||||
"name": "duration_sec",
|
||||
"type": "float",
|
||||
"description": "Explicit output duration in seconds; must be positive when set, otherwise predicted duration is used.",
|
||||
"required": false,
|
||||
"min": 0.0
|
||||
},
|
||||
{
|
||||
"name": "duration_scale",
|
||||
"type": "float",
|
||||
"description": "Predicted-duration multiplier; must be positive when set; default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Text chunking mode; default endline.",
|
||||
"values": [
|
||||
"japanese",
|
||||
"endline"
|
||||
],
|
||||
"required": false,
|
||||
"default": "endline"
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_size",
|
||||
"type": "int",
|
||||
"description": "Maximum characters per text chunk; default uses the model text-token window.",
|
||||
"required": false,
|
||||
"min": 1
|
||||
},
|
||||
{
|
||||
"name": "min_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Minimum generated duration in seconds; must be positive and no greater than max_duration_sec; default 0.5.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.5
|
||||
},
|
||||
{
|
||||
"name": "max_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Maximum generated duration in seconds; must be at least min_duration_sec; default 30.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 30.0
|
||||
},
|
||||
{
|
||||
"name": "text_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Text classifier-free guidance scale; default 3.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 3.0
|
||||
},
|
||||
{
|
||||
"name": "speaker_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Speaker classifier-free guidance scale; default 5.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 5.0
|
||||
},
|
||||
{
|
||||
"name": "caption_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Caption classifier-free guidance scale; default 3.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 3.0
|
||||
},
|
||||
{
|
||||
"name": "guidance_scale",
|
||||
"type": "float",
|
||||
"description": "Override all classifier-free guidance scales when set.",
|
||||
"required": false,
|
||||
"min": 0.0
|
||||
},
|
||||
{
|
||||
"name": "guidance_mode",
|
||||
"type": "enum",
|
||||
"description": "Classifier-free guidance combination mode; default independent.",
|
||||
"values": [
|
||||
"independent",
|
||||
"joint",
|
||||
"alternating"
|
||||
],
|
||||
"required": false,
|
||||
"default": "independent"
|
||||
},
|
||||
{
|
||||
"name": "guidance_min_t",
|
||||
"type": "float",
|
||||
"description": "Minimum diffusion timestep value where guidance is active; default 0.5.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.5
|
||||
},
|
||||
{
|
||||
"name": "guidance_max_t",
|
||||
"type": "float",
|
||||
"description": "Maximum diffusion timestep value where guidance is active; default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Generation seed for reproducible output; omitted uses a random seed.",
|
||||
"required": false,
|
||||
"min": 0
|
||||
},
|
||||
{
|
||||
"name": "trim_tail",
|
||||
"type": "bool",
|
||||
"description": "Trim trailing silence-like samples; default true.",
|
||||
"required": false,
|
||||
"default": true
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Model weight storage type; default native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "codec_weight_type",
|
||||
"type": "enum",
|
||||
"description": "DACVAE codec weight storage type; default native.",
|
||||
"preset": "weight_type_codec_q8",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "condition_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Condition encoder graph arena size in MiB; default 256.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 256
|
||||
},
|
||||
{
|
||||
"name": "rf_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "RF sampler graph arena size in MiB; default 768.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 768
|
||||
},
|
||||
{
|
||||
"name": "codec_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "DACVAE codec graph arena size in MiB; default 512.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 512
|
||||
},
|
||||
{
|
||||
"name": "condition_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Condition encoder weight context size in MiB; default 32.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 32
|
||||
},
|
||||
{
|
||||
"name": "rf_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "RF sampler weight context size in MiB; default 32.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 32
|
||||
},
|
||||
{
|
||||
"name": "codec_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "DACVAE codec weight context size in MiB; default 32.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 32
|
||||
},
|
||||
{
|
||||
"name": "mem_saver",
|
||||
"type": "bool",
|
||||
"description": "Release staged runtime graphs after request phases; default true.",
|
||||
"required": false,
|
||||
"default": true
|
||||
},
|
||||
{
|
||||
"name": "reference_cache_slots",
|
||||
"type": "int",
|
||||
"description": "Prepared reference-speaker cache slots; default 1.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 1
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"dependencies": [],
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "irodori_tts_v4_small_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"Design",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "irodori_tts_v4_small_q8_0",
|
||||
"display_name": "Irodori-TTS v4 Small Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Irodori-TTS-v4-Small-GGUF",
|
||||
"files": [
|
||||
"Irodori-TTS-v4-Small-GGUF/irodori-tts-v4-small-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Irodori-TTS-v4-Small-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "irodori_tts_v4_small_f16",
|
||||
"display_name": "Irodori-TTS v4 Small F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Irodori-TTS-v4-Small-GGUF",
|
||||
"files": [
|
||||
"Irodori-TTS-v4-Small-GGUF/irodori-tts-v4-small-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Irodori-TTS-v4-Small-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "irodori_tts_600m_v3_voicedesign_q8_0",
|
||||
"display_name": "Irodori-TTS 600M v3 VoiceDesign Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Irodori-TTS-600M-v3-VoiceDesign-GGUF",
|
||||
"files": [
|
||||
"Irodori-TTS-600M-v3-VoiceDesign-GGUF/irodori-tts-600m-v3-voicedesign-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Irodori-TTS-600M-v3-VoiceDesign-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "irodori_tts_600m_v3_voicedesign_f16",
|
||||
"display_name": "Irodori-TTS 600M v3 VoiceDesign F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Irodori-TTS-600M-v3-VoiceDesign-GGUF",
|
||||
"files": [
|
||||
"Irodori-TTS-600M-v3-VoiceDesign-GGUF/irodori-tts-600m-v3-voicedesign-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Irodori-TTS-600M-v3-VoiceDesign-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "irodori_tts_500m_v3_q8_0",
|
||||
"display_name": "Irodori-TTS 500M v3 Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Irodori-TTS-500M-v3-GGUF",
|
||||
"files": [
|
||||
"Irodori-TTS-500M-v3-GGUF/irodori-tts-500m-v3-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Irodori-TTS-500M-v3-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "irodori_tts_500m_v3_f16",
|
||||
"display_name": "Irodori-TTS 500M v3 F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Irodori-TTS-500M-v3-GGUF",
|
||||
"files": [
|
||||
"Irodori-TTS-500M-v3-GGUF/irodori-tts-500m-v3-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Irodori-TTS-500M-v3-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"model_config": "model:model_config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_v4_json": "model:tokenizer/tokenizer.json",
|
||||
"pretrained_text_config": "model:text_encoder_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"codec_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "codec_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"tokenizer": "../llm-jp-3-150m",
|
||||
"codec": "../Semantic-DACVAE-Japanese-32dim"
|
||||
},
|
||||
"files": {
|
||||
"model_config": "model:model_config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"tokenizer_json": "tokenizer:tokenizer.json",
|
||||
"tokenizer_v4_json": "model:tokenizer/tokenizer.json",
|
||||
"pretrained_text_config": "model:text_encoder_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors",
|
||||
"codec_weights": "codec:weights.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "kroko_asr",
|
||||
"display_name": "Kroko Community ASR",
|
||||
"description": "Native Zipformer2 RNN-T transcription for the public free Kroko Community single-language packages, with offline and stateful streaming inference, greedy and modified beam decoding, hotwords, endpoint segments, partial results, and word timestamps.",
|
||||
"category": "asr",
|
||||
"status": "community",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"de",
|
||||
"en",
|
||||
"es",
|
||||
"fr",
|
||||
"it",
|
||||
"he",
|
||||
"nl",
|
||||
"pt",
|
||||
"sv",
|
||||
"tr"
|
||||
],
|
||||
"capabilities": {
|
||||
"asr": [
|
||||
"word_timestamps",
|
||||
"partial_results",
|
||||
"segments"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "language",
|
||||
"type": "string",
|
||||
"description": "Language code matching the selected single-language Kroko package; default auto uses the package language.",
|
||||
"required": false,
|
||||
"default": "auto"
|
||||
},
|
||||
{
|
||||
"name": "decoding_method",
|
||||
"type": "enum",
|
||||
"description": "RNN-T decoding method; greedy search remains the parity-tested default.",
|
||||
"values": [
|
||||
"greedy_search",
|
||||
"modified_beam_search"
|
||||
],
|
||||
"required": false,
|
||||
"default": "greedy_search"
|
||||
},
|
||||
{
|
||||
"name": "num_beams",
|
||||
"type": "int",
|
||||
"description": "Maximum active hypotheses retained by modified beam search.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"max": 64,
|
||||
"default": 4
|
||||
},
|
||||
{
|
||||
"name": "blank_penalty",
|
||||
"type": "float",
|
||||
"description": "Non-negative score subtracted from the RNN-T blank logit.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.0
|
||||
},
|
||||
{
|
||||
"name": "hotwords",
|
||||
"type": "string",
|
||||
"description": "Slash- or newline-separated natural-text phrases; requires modified beam search.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "hotwords_score",
|
||||
"type": "float",
|
||||
"description": "Non-negative per-token context boost for hotword phrases.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.5
|
||||
},
|
||||
{
|
||||
"name": "enable_endpoint",
|
||||
"type": "bool",
|
||||
"description": "Enable automatic endpoint speech segments.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "rule1_min_trailing_silence_sec",
|
||||
"type": "float",
|
||||
"description": "Endpoint timeout in seconds even when no speech token was decoded.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 2.4
|
||||
},
|
||||
{
|
||||
"name": "rule2_min_trailing_silence_sec",
|
||||
"type": "float",
|
||||
"description": "Endpoint silence in seconds after a speech token was decoded.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.2
|
||||
},
|
||||
{
|
||||
"name": "rule3_min_utterance_length_sec",
|
||||
"type": "float",
|
||||
"description": "Maximum utterance duration in seconds before an endpoint.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 20.0
|
||||
}
|
||||
],
|
||||
"session": [],
|
||||
"load": []
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokens": "model:tokens.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokens": "model:tokens.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
],
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "kroko_asr_community_q8_0",
|
||||
"display_name": "Kroko Community ASR Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Kroko-ASR-GGUF",
|
||||
"files": [
|
||||
"Kroko-ASR-GGUF/kroko-en-community-64-l-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Kroko-ASR-GGUF"
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "kroko_asr_community_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/community_models/kroko_asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,125 @@
|
||||
{
|
||||
"family": "miotts",
|
||||
"display_name": "MioTTS",
|
||||
"description": "Lightweight LLM-based English and Japanese TTS family built on MioCodec, supporting low-latency speech generation and zero-shot voice cloning from short reference audio.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en",
|
||||
"ja"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "miotts_1_7b_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "miotts_1_7b_q8_0",
|
||||
"display_name": "MioTTS 1.7B Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "MioTTS-1.7B-GGUF",
|
||||
"files": [
|
||||
"MioTTS-1.7B-GGUF/miotts-1.7b-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "MioTTS-1.7B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "miotts_1_7b_bf16",
|
||||
"display_name": "MioTTS 1.7B BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "MioTTS-1.7B-GGUF",
|
||||
"files": [
|
||||
"MioTTS-1.7B-GGUF/miotts-1.7b-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "MioTTS-1.7B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "miotts_1_7b_orig",
|
||||
"display_name": "MioTTS 1.7B Original-Dtype GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "MioTTS-1.7B-GGUF",
|
||||
"files": [
|
||||
"MioTTS-1.7B-GGUF/miotts-1.7b-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "MioTTS-1.7B-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt"
|
||||
},
|
||||
"optional_files": {
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt"
|
||||
},
|
||||
"optional_files": {
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
{
|
||||
"family": "moss_tts_local",
|
||||
"display_name": "MOSS-TTS-Local",
|
||||
"description": "Flagship MOSS-TTS model for high-fidelity 31-language and code-switched speech, zero-shot voice cloning, long-form generation, and fine-grained Pinyin, phoneme, and duration control.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"ar",
|
||||
"cs",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"en",
|
||||
"es",
|
||||
"fa",
|
||||
"fi",
|
||||
"fr",
|
||||
"he",
|
||||
"hi",
|
||||
"hu",
|
||||
"it",
|
||||
"ja",
|
||||
"ko",
|
||||
"mk",
|
||||
"ms",
|
||||
"nl",
|
||||
"pl",
|
||||
"pt",
|
||||
"ro",
|
||||
"ru",
|
||||
"sv",
|
||||
"sw",
|
||||
"th",
|
||||
"tl",
|
||||
"tr",
|
||||
"vi",
|
||||
"yue",
|
||||
"zh"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "moss_tts_local_v1_5_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/moss_tts.md",
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "moss_tts_local_v1_5_q8_0",
|
||||
"display_name": "MOSS-TTS-Local v1.5 Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "MOSS-TTS-Local-v1.5-GGUF",
|
||||
"files": [
|
||||
"MOSS-TTS-Local-v1.5-GGUF/moss-tts-local-v1.5-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "MOSS-TTS-Local-v1.5-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "moss_tts_local_v1_5_bf16",
|
||||
"display_name": "MOSS-TTS-Local v1.5 BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "MOSS-TTS-Local-v1.5-GGUF",
|
||||
"files": [
|
||||
"MOSS-TTS-Local-v1.5-GGUF/moss-tts-local-v1.5-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "MOSS-TTS-Local-v1.5-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_vocab": "model:vocab.json",
|
||||
"tokenizer_merges": "model:merges.txt",
|
||||
"audio_tokenizer_config": "model:audio_tokenizer/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"audio_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "audio_tokenizer_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"audio_tokenizer": "audio_tokenizer"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_vocab": "model:vocab.json",
|
||||
"tokenizer_merges": "model:merges.txt",
|
||||
"audio_tokenizer_config": "audio_tokenizer:config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors",
|
||||
"audio_tokenizer_weights": "audio_tokenizer:model.safetensors.index.json"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
{
|
||||
"family": "moss_tts_nano",
|
||||
"display_name": "MOSS-TTS-Nano",
|
||||
"description": "Compact deployment-first MOSS-TTS model for real-time multilingual speech generation, lightweight integration, and zero-shot voice cloning.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"ar",
|
||||
"cs",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"en",
|
||||
"es",
|
||||
"fa",
|
||||
"fr",
|
||||
"hu",
|
||||
"it",
|
||||
"ja",
|
||||
"ko",
|
||||
"pl",
|
||||
"pt",
|
||||
"ru",
|
||||
"sv",
|
||||
"tr",
|
||||
"zh"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "moss_tts_nano_100m_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/moss_tts.md",
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "moss_tts_nano_100m_q8_0",
|
||||
"display_name": "MOSS-TTS-Nano 100M Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "MOSS-TTS-Nano-100M-GGUF",
|
||||
"files": [
|
||||
"MOSS-TTS-Nano-100M-GGUF/moss-tts-nano-100m-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "MOSS-TTS-Nano-100M-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "moss_tts_nano_100m_bf16",
|
||||
"display_name": "MOSS-TTS-Nano 100M BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "MOSS-TTS-Nano-100M-GGUF",
|
||||
"files": [
|
||||
"MOSS-TTS-Nano-100M-GGUF/moss-tts-nano-100m-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "MOSS-TTS-Nano-100M-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_model": "model:tokenizer.model",
|
||||
"audio_tokenizer_config": "model:audio_tokenizer/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"audio_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "audio_tokenizer_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"audio_tokenizer": "audio_tokenizer"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_model": "model:tokenizer.model",
|
||||
"audio_tokenizer_config": "audio_tokenizer:config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors",
|
||||
"audio_tokenizer_weights": "audio_tokenizer:model.safetensors.index.json"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,156 @@
|
||||
{
|
||||
"family": "nemotron_asr",
|
||||
"display_name": "Nemotron 3.5 ASR",
|
||||
"description": "NVIDIA 600M streaming ASR model for low-latency and batch transcription across 40 language-locales, with native punctuation, capitalization, automatic language detection, and configurable chunk sizes.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"ar-AR",
|
||||
"bg-BG",
|
||||
"cs-CZ",
|
||||
"da-DK",
|
||||
"de-DE",
|
||||
"el-GR",
|
||||
"en-GB",
|
||||
"en-US",
|
||||
"es-ES",
|
||||
"es-US",
|
||||
"et-EE",
|
||||
"fi-FI",
|
||||
"fr-CA",
|
||||
"fr-FR",
|
||||
"he-IL",
|
||||
"hi-IN",
|
||||
"hr-HR",
|
||||
"hu-HU",
|
||||
"it-IT",
|
||||
"ja-JP",
|
||||
"ko-KR",
|
||||
"lt-LT",
|
||||
"lv-LV",
|
||||
"mt-MT",
|
||||
"nb-NO",
|
||||
"nl-NL",
|
||||
"nn-NO",
|
||||
"pl-PL",
|
||||
"pt-BR",
|
||||
"pt-PT",
|
||||
"ro-RO",
|
||||
"ru-RU",
|
||||
"sk-SK",
|
||||
"sl-SI",
|
||||
"sv-SE",
|
||||
"th-TH",
|
||||
"tr-TR",
|
||||
"uk-UA",
|
||||
"vi-VN",
|
||||
"zh-CN"
|
||||
],
|
||||
"capabilities": {},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "nemotron_asr_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "nemotron_asr_q8_0",
|
||||
"display_name": "Nemotron 3.5 ASR Streaming 0.6B Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF",
|
||||
"files": [
|
||||
"Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "nemotron_asr_f16",
|
||||
"display_name": "Nemotron 3.5 ASR Streaming 0.6B F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF",
|
||||
"files": [
|
||||
"Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "nemotron_asr_safetensors",
|
||||
"display_name": "Nemotron 3.5 ASR Streaming 0.6B Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "nemotron-3.5-asr-streaming-0.6b",
|
||||
"files": [
|
||||
"config.json",
|
||||
"model.safetensors",
|
||||
"processor_config.json",
|
||||
"tokenizer.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "nvidia/nemotron-3.5-asr-streaming-0.6b"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,157 @@
|
||||
{
|
||||
"family": "omnivoice",
|
||||
"display_name": "OmniVoice",
|
||||
"description": "Massively multilingual zero-shot TTS model from k2-fsa for 600+ languages, supporting short-reference voice cloning, attribute-based voice design, pronunciation controls, and nonverbal tags.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone",
|
||||
"design"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"600+ languages"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"design": [
|
||||
"voice_design"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "omnivoice_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"Design",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/omnivoice.md",
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "omnivoice_q8_0",
|
||||
"display_name": "OmniVoice Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "OmniVoice-GGUF",
|
||||
"files": [
|
||||
"OmniVoice-GGUF/omnivoice-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "OmniVoice-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "omnivoice_bf16",
|
||||
"display_name": "OmniVoice BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "OmniVoice-GGUF",
|
||||
"files": [
|
||||
"OmniVoice-GGUF/omnivoice-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "OmniVoice-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "omnivoice_f16",
|
||||
"display_name": "OmniVoice F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "OmniVoice-GGUF",
|
||||
"files": [
|
||||
"OmniVoice-GGUF/omnivoice-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "OmniVoice-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "omnivoice_safetensors",
|
||||
"display_name": "OmniVoice Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "OmniVoice",
|
||||
"files": [
|
||||
"config.json",
|
||||
"model.safetensors",
|
||||
"tokenizer.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "k2-fsa/OmniVoice"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"audio_tokenizer_config": "model:audio_tokenizer/config.json",
|
||||
"audio_tokenizer_preprocessor": "model:audio_tokenizer/preprocessor_config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"chat_template": "model:chat_template.jinja"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "weights"
|
||||
},
|
||||
"audio_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "audio_tokenizer_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"audio_tokenizer_config": "model:audio_tokenizer/config.json",
|
||||
"audio_tokenizer_preprocessor": "model:audio_tokenizer/preprocessor_config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"chat_template": "model:chat_template.jinja"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors",
|
||||
"audio_tokenizer_weights": "model:audio_tokenizer/model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,300 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "outetts",
|
||||
"display_name": "Llama-OuteTTS 1.0",
|
||||
"description": "Llama-based open-weight TTS model for 23-language speech synthesis with one-shot voice cloning from short reference audio and automatic word-alignment support.",
|
||||
"category": "tts",
|
||||
"status": "community",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"ar",
|
||||
"be",
|
||||
"bn",
|
||||
"de",
|
||||
"en",
|
||||
"es",
|
||||
"fa",
|
||||
"fr",
|
||||
"hu",
|
||||
"it",
|
||||
"ja",
|
||||
"ka",
|
||||
"ko",
|
||||
"lt",
|
||||
"lv",
|
||||
"nl",
|
||||
"pl",
|
||||
"pt",
|
||||
"ru",
|
||||
"sw",
|
||||
"ta",
|
||||
"uk",
|
||||
"zh"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum generated audio tokens per chunk. When omitted, OuteTTS estimates a safe value from each chunk.",
|
||||
"required": false,
|
||||
"min": 1
|
||||
},
|
||||
{
|
||||
"name": "temperature",
|
||||
"type": "float",
|
||||
"description": "Sampling temperature; default 0.4 for cloning, otherwise model config default.",
|
||||
"required": false,
|
||||
"min": 0.0
|
||||
},
|
||||
{
|
||||
"name": "top_k",
|
||||
"type": "int",
|
||||
"description": "Top-k sampling; default 40 for cloning, otherwise model config default.",
|
||||
"required": false,
|
||||
"min": 0
|
||||
},
|
||||
{
|
||||
"name": "top_p",
|
||||
"type": "float",
|
||||
"description": "Nucleus sampling in (0, 1]; default 0.9 for cloning, otherwise model config default.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"max": 1.0
|
||||
},
|
||||
{
|
||||
"name": "min_p",
|
||||
"type": "float",
|
||||
"description": "Minimum probability relative to the best token; default 0.05 for cloning, otherwise model config default.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"max": 1.0
|
||||
},
|
||||
{
|
||||
"name": "repetition_penalty",
|
||||
"type": "float",
|
||||
"description": "Positive windowed repetition penalty; default 1.1.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.1
|
||||
},
|
||||
{
|
||||
"name": "repetition_window",
|
||||
"type": "int",
|
||||
"description": "Recent-token penalty window; default 64.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 64
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Sampling seed; cloning defaults to 4099 for native weights and 42 for quantized weights.",
|
||||
"required": false,
|
||||
"min": 0
|
||||
},
|
||||
{
|
||||
"name": "reference_text",
|
||||
"type": "string",
|
||||
"description": "Transcript matching the reference voice audio for voice cloning.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "reference_language",
|
||||
"type": "string",
|
||||
"description": "Language code used to align the reference transcript; default en.",
|
||||
"required": false,
|
||||
"default": "en"
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_size",
|
||||
"type": "int",
|
||||
"description": "Maximum UTF-8 codepoints per long-form text chunk; default 256. Chunks are split further when required by max_tokens or context budget.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 256
|
||||
},
|
||||
{
|
||||
"name": "text_chunk_mode",
|
||||
"type": "enum",
|
||||
"description": "Framework long-form text chunking mode; default word_budget.",
|
||||
"preset": "text_chunk_mode_full",
|
||||
"required": false,
|
||||
"default": "word_budget"
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Language-model weight storage type. Quantized CUDA voice cloning is expanded to F32 in memory for generation correctness.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "llama_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Language-model weight context size in MiB; default 4096.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 4096
|
||||
},
|
||||
{
|
||||
"name": "constant_context_mb",
|
||||
"type": "int",
|
||||
"description": "Language-model constant tensor context size in MiB; default 256.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 256
|
||||
},
|
||||
{
|
||||
"name": "dac_weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "DAC decoder weight context size in MiB; default 1024.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1024
|
||||
},
|
||||
{
|
||||
"name": "dac_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "DAC decoder graph arena size in MiB; default 1536.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1536
|
||||
},
|
||||
{
|
||||
"name": "aligner_path",
|
||||
"type": "path",
|
||||
"description": "Optional Qwen3 Forced Aligner override. Cloning automatically uses the aligner embedded in a standalone OuteTTS GGUF when present.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "reference_cache_slots",
|
||||
"type": "int",
|
||||
"description": "Prepared reference-profile cache slots; default 1, set 0 to disable.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 1
|
||||
},
|
||||
{
|
||||
"name": "mem_saver",
|
||||
"type": "bool",
|
||||
"description": "Release cached-step and aligner runtime state after use; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json",
|
||||
"dac_config": "model:dac/config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"aligner_config": "model:aligner/config.json",
|
||||
"aligner_generation_config": "model:aligner/generation_config.json",
|
||||
"aligner_tokenizer_config": "model:aligner/tokenizer_config.json",
|
||||
"aligner_preprocessor_config": "model:aligner/preprocessor_config.json",
|
||||
"aligner_processor_config": "model:aligner/processor_config.json",
|
||||
"aligner_chat_template": "model:aligner/chat_template.json",
|
||||
"aligner_chat_template_jinja": "model:aligner/chat_template.jinja",
|
||||
"aligner_vocab": "model:aligner/vocab.json",
|
||||
"aligner_merges": "model:aligner/merges.txt",
|
||||
"aligner_tokenizer_json": "model:aligner/tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"dac_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "dac_weights"
|
||||
},
|
||||
"aligner_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "aligner_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"dac": "../DAC.speech.v1.0"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer": "model:tokenizer.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json",
|
||||
"dac_config": "dac:config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors",
|
||||
"dac_weights": "dac:model.safetensors"
|
||||
}
|
||||
}
|
||||
],
|
||||
"packages": [
|
||||
{
|
||||
"id": "outetts_1_0_1b_q8_0",
|
||||
"display_name": "Llama-OuteTTS 1.0 1B Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Llama-OuteTTS-1.0-1B_Q8",
|
||||
"files": [
|
||||
"Text to audio (TTS)/Llama-OuteTTS-1.0-1B_Q8.gguf"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "mirek190/audio.cpp"
|
||||
}
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "outetts_1_0_1b_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/community_models/outetts.md",
|
||||
"docs/reports/outetts_validation.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,260 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"family": "parakeet_tdt",
|
||||
"display_name": "Parakeet-TDT 0.6B v3",
|
||||
"description": "NVIDIA Parakeet-TDT 0.6B v3 FastConformer-TDT ASR covering 25 European languages with automatic language detection. Supports the upstream Transformers-compatible safetensors package and standalone audio.cpp GGUF, with offline full-context, bounded-window long-form, and buffered streaming; the checkpoint uses unlimited bidirectional attention and is not a native cache-aware streaming model.",
|
||||
"category": "asr",
|
||||
"status": "community",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"bg",
|
||||
"cs",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"en",
|
||||
"es",
|
||||
"et",
|
||||
"fi",
|
||||
"fr",
|
||||
"hr",
|
||||
"hu",
|
||||
"it",
|
||||
"lt",
|
||||
"lv",
|
||||
"mt",
|
||||
"nl",
|
||||
"pl",
|
||||
"pt",
|
||||
"ro",
|
||||
"ru",
|
||||
"sk",
|
||||
"sl",
|
||||
"sv",
|
||||
"uk"
|
||||
],
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"capabilities": {
|
||||
"asr": [
|
||||
"word_timestamps",
|
||||
"partial_results"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "max_tokens",
|
||||
"type": "int",
|
||||
"description": "Maximum TDT generated tokens; 0 or omitted uses the model-derived limit.",
|
||||
"required": false,
|
||||
"min": 0,
|
||||
"default": 0
|
||||
},
|
||||
{
|
||||
"name": "keep_language_tags",
|
||||
"type": "bool",
|
||||
"description": "Keep language tag tokens in decoded text; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Shared matmul weight storage type; default native.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "matmul_weight_type",
|
||||
"type": "enum",
|
||||
"description": "Encoder and decoder matmul weight storage type; defaults to weight_type, which defaults to native. Q8_0 measured 1.79x faster on the tested CPU and changed roughly 8 percent of transcripts without moving aggregate word error rate.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "conv_weight_type",
|
||||
"type": "enum",
|
||||
"description": "Convolution weight storage type; default native.",
|
||||
"preset": "weight_type_conv",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
},
|
||||
{
|
||||
"name": "perf_mode",
|
||||
"type": "enum",
|
||||
"description": "Encoder attention implementation. Default off uses the validated relative-attention path; flash_attention enables the fused implementation, which was numerically validated but slower on the tested hardware.",
|
||||
"preset": "perf_mode_flash_attention",
|
||||
"required": false,
|
||||
"default": "off"
|
||||
},
|
||||
{
|
||||
"name": "weight_context_mb",
|
||||
"type": "int",
|
||||
"description": "Weight context arena size in MiB; default 3072.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 3072
|
||||
},
|
||||
{
|
||||
"name": "encoder_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Encoder graph arena size in MiB; default 1024.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 1024
|
||||
},
|
||||
{
|
||||
"name": "decoder_graph_arena_mb",
|
||||
"type": "int",
|
||||
"description": "Decoder graph arena size in MiB; default 256.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 256
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_duration_sec",
|
||||
"type": "float",
|
||||
"description": "Center-region duration for buffered streaming in seconds; default 2. Fixed context windows are re-encoded rather than cache-aware.",
|
||||
"required": false,
|
||||
"min": 0.001,
|
||||
"default": 2.0
|
||||
},
|
||||
{
|
||||
"name": "left_context_sec",
|
||||
"type": "float",
|
||||
"description": "Past context included when re-encoding each buffered-streaming window in seconds; default 10.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 10.0
|
||||
},
|
||||
{
|
||||
"name": "right_context_sec",
|
||||
"type": "float",
|
||||
"description": "Future lookahead included when re-encoding each buffered-streaming window in seconds; default 2 and adds equivalent partial-result latency.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 2.0
|
||||
},
|
||||
{
|
||||
"name": "streaming_attention_mode",
|
||||
"type": "enum",
|
||||
"description": "Attention policy inside each buffered window. full_context preserves bidirectional attention over the bounded window.",
|
||||
"values": [
|
||||
"full_context"
|
||||
],
|
||||
"required": false,
|
||||
"default": "full_context"
|
||||
},
|
||||
{
|
||||
"name": "offline_mode",
|
||||
"type": "enum",
|
||||
"description": "Offline encoder scheduling. full_context encodes the whole utterance, long_form uses bounded overlapping windows, and auto selects long_form beyond audio_chunk_threshold_sec.",
|
||||
"values": [
|
||||
"full_context",
|
||||
"long_form",
|
||||
"auto"
|
||||
],
|
||||
"required": false,
|
||||
"default": "full_context"
|
||||
},
|
||||
{
|
||||
"name": "audio_chunk_threshold_sec",
|
||||
"type": "float",
|
||||
"description": "Duration threshold used by offline_mode=auto before switching to bounded-window long-form execution; default 30 seconds.",
|
||||
"required": false,
|
||||
"min": 0.001,
|
||||
"default": 30.0
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "parakeet_tdt_q8_0",
|
||||
"display_name": "Parakeet-TDT 0.6B v3 Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Parakeet-TDT-0.6B-v3-GGUF",
|
||||
"files": [
|
||||
"Parakeet-TDT-0.6B-v3-GGUF/parakeet-tdt-0.6b-v3-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Parakeet-TDT-0.6B-v3-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "parakeet_tdt_f16",
|
||||
"display_name": "Parakeet-TDT 0.6B v3 F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Parakeet-TDT-0.6B-v3-GGUF",
|
||||
"files": [
|
||||
"Parakeet-TDT-0.6B-v3-GGUF/parakeet-tdt-0.6b-v3-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Parakeet-TDT-0.6B-v3-GGUF"
|
||||
}
|
||||
],
|
||||
"dependencies": [],
|
||||
"ui": {
|
||||
"recommended_package": "parakeet_tdt_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/community_models/parakeet_tdt.md"
|
||||
]
|
||||
},
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,238 @@
|
||||
{
|
||||
"family": "pocket_tts",
|
||||
"display_name": "PocketTTS",
|
||||
"description": "Kyutai 100M-parameter CPU-friendly TTS package set for real-time local synthesis and small-footprint voice cloning in English, German, Italian, Portuguese, and Spanish.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en",
|
||||
"de",
|
||||
"it",
|
||||
"pt",
|
||||
"es"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "pocket_tts_english_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "pocket_tts_english_q8_0",
|
||||
"display_name": "PocketTTS English Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "PocketTTS-GGUF/english",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/english/pocket-tts-english-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/english"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_english_bf16",
|
||||
"display_name": "PocketTTS English BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "PocketTTS-GGUF/english",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/english/pocket-tts-english-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/english"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_german_q8_0",
|
||||
"display_name": "PocketTTS German Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "PocketTTS-GGUF/german",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/german/pocket-tts-german-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/german"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_german_bf16",
|
||||
"display_name": "PocketTTS German BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "PocketTTS-GGUF/german",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/german/pocket-tts-german-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/german"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_italian_q8_0",
|
||||
"display_name": "PocketTTS Italian Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "PocketTTS-GGUF/italian",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/italian/pocket-tts-italian-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/italian"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_italian_bf16",
|
||||
"display_name": "PocketTTS Italian BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "PocketTTS-GGUF/italian",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/italian/pocket-tts-italian-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/italian"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_portuguese_q8_0",
|
||||
"display_name": "PocketTTS Portuguese Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "PocketTTS-GGUF/portuguese",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/portuguese/pocket-tts-portuguese-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/portuguese"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_portuguese_bf16",
|
||||
"display_name": "PocketTTS Portuguese BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "PocketTTS-GGUF/portuguese",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/portuguese/pocket-tts-portuguese-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/portuguese"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_spanish_q8_0",
|
||||
"display_name": "PocketTTS Spanish Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "PocketTTS-GGUF/spanish",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/spanish/pocket-tts-spanish-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/spanish"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_spanish_bf16",
|
||||
"display_name": "PocketTTS Spanish BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "PocketTTS-GGUF/spanish",
|
||||
"files": [
|
||||
"PocketTTS-GGUF/spanish/pocket-tts-spanish-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "PocketTTS-GGUF/spanish"
|
||||
},
|
||||
{
|
||||
"id": "pocket_tts_english_safetensors",
|
||||
"display_name": "PocketTTS English Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "pocket-tts",
|
||||
"files": [
|
||||
"languages/english/embeddings/alba.safetensors",
|
||||
"languages/english/embeddings/anna.safetensors",
|
||||
"languages/english/embeddings/azelma.safetensors",
|
||||
"languages/english/embeddings/bill_boerst.safetensors",
|
||||
"languages/english/embeddings/caro_davy.safetensors",
|
||||
"languages/english/embeddings/charles.safetensors",
|
||||
"languages/english/embeddings/cosette.safetensors",
|
||||
"languages/english/embeddings/eponine.safetensors",
|
||||
"languages/english/embeddings/estelle.safetensors",
|
||||
"languages/english/embeddings/eve.safetensors",
|
||||
"languages/english/embeddings/fantine.safetensors",
|
||||
"languages/english/embeddings/george.safetensors",
|
||||
"languages/english/embeddings/giovanni.safetensors",
|
||||
"languages/english/embeddings/jane.safetensors",
|
||||
"languages/english/embeddings/javert.safetensors",
|
||||
"languages/english/embeddings/jean.safetensors",
|
||||
"languages/english/embeddings/juergen.safetensors",
|
||||
"languages/english/embeddings/lola.safetensors",
|
||||
"languages/english/embeddings/marius.safetensors",
|
||||
"languages/english/embeddings/mary.safetensors",
|
||||
"languages/english/embeddings/michael.safetensors",
|
||||
"languages/english/embeddings/paul.safetensors",
|
||||
"languages/english/embeddings/peter_yearsley.safetensors",
|
||||
"languages/english/embeddings/rafael.safetensors",
|
||||
"languages/english/embeddings/stuart_bell.safetensors",
|
||||
"languages/english/embeddings/vera.safetensors",
|
||||
"languages/english/model.safetensors",
|
||||
"languages/english/tokenizer.model"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "kyutai/pocket-tts"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"tokenizer": "model:tokenizer.model"
|
||||
},
|
||||
"optional_files": {
|
||||
"config": "model:config.yaml"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"language": "languages/english"
|
||||
},
|
||||
"files": {
|
||||
"tokenizer": "language:tokenizer.model"
|
||||
},
|
||||
"optional_files": {
|
||||
"config": "language:config.yaml"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "language:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,214 @@
|
||||
{
|
||||
"family": "qwen3_asr",
|
||||
"display_name": "Qwen3-ASR",
|
||||
"description": "Qwen ASR model family for language identification and speech recognition across 30 languages, 22 Chinese dialects, and multiple English accents, with robustness for noisy, long-form, and singing audio.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"zh",
|
||||
"en",
|
||||
"yue",
|
||||
"ar",
|
||||
"de",
|
||||
"fr",
|
||||
"es",
|
||||
"pt",
|
||||
"id",
|
||||
"it",
|
||||
"ko",
|
||||
"ru",
|
||||
"th",
|
||||
"vi",
|
||||
"ja",
|
||||
"tr",
|
||||
"hi",
|
||||
"ms",
|
||||
"nl",
|
||||
"sv",
|
||||
"da",
|
||||
"fi",
|
||||
"pl",
|
||||
"cs",
|
||||
"fil",
|
||||
"fa",
|
||||
"el",
|
||||
"hu",
|
||||
"mk",
|
||||
"ro",
|
||||
"zh dialects"
|
||||
],
|
||||
"capabilities": {
|
||||
"asr": [
|
||||
"word_timestamps",
|
||||
"vad_chunking",
|
||||
"partial_results"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "qwen3_asr_1_7b_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/qwen3.md",
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "qwen3_asr_1_7b_q8_0",
|
||||
"display_name": "Qwen3-ASR 1.7B Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Qwen3-ASR-1.7B-GGUF",
|
||||
"files": [
|
||||
"Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-ASR-1.7B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_asr_1_7b_f16",
|
||||
"display_name": "Qwen3-ASR 1.7B F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Qwen3-ASR-1.7B-GGUF",
|
||||
"files": [
|
||||
"Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-ASR-1.7B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_asr_0_6b_q8_0",
|
||||
"display_name": "Qwen3-ASR 0.6B Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Qwen3-ASR-0.6B-GGUF",
|
||||
"files": [
|
||||
"Qwen3-ASR-0.6B-GGUF/qwen3-asr-0.6b-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-ASR-0.6B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_asr_0_6b_f16",
|
||||
"display_name": "Qwen3-ASR 0.6B F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Qwen3-ASR-0.6B-GGUF",
|
||||
"files": [
|
||||
"Qwen3-ASR-0.6B-GGUF/qwen3-asr-0.6b-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-ASR-0.6B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_asr_1_7b_safetensors",
|
||||
"display_name": "Qwen3-ASR 1.7B HF Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "Qwen3-ASR-1.7B-hf",
|
||||
"files": [
|
||||
"config.json",
|
||||
"generation_config.json",
|
||||
"model.safetensors",
|
||||
"processor_config.json",
|
||||
"tokenizer_config.json",
|
||||
"tokenizer.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "Qwen/Qwen3-ASR-1.7B-hf"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "qwen3_asr_0_6b_safetensors",
|
||||
"display_name": "Qwen3-ASR 0.6B Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "Qwen3-ASR-0.6B",
|
||||
"files": [
|
||||
"config.json",
|
||||
"generation_config.json",
|
||||
"model.safetensors",
|
||||
"preprocessor_config.json",
|
||||
"tokenizer_config.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "Qwen/Qwen3-ASR-0.6B"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"preprocessor_config": "model:preprocessor_config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"chat_template": "model:chat_template.json",
|
||||
"chat_template_jinja": "model:chat_template.jinja",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"preprocessor_config": "model:preprocessor_config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"chat_template": "model:chat_template.json",
|
||||
"chat_template_jinja": "model:chat_template.jinja",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt",
|
||||
"tokenizer_json": "model:tokenizer.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,225 @@
|
||||
{
|
||||
"family": "qwen3_tts",
|
||||
"display_name": "Qwen3-TTS",
|
||||
"description": "Qwen TTS family for controllable 10-language speech synthesis, including 3-second voice cloning, CustomVoice instruction control over preset timbres, and VoiceDesign from natural-language descriptions.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone",
|
||||
"design"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"zh",
|
||||
"en",
|
||||
"ja",
|
||||
"ko",
|
||||
"de",
|
||||
"fr",
|
||||
"ru",
|
||||
"pt",
|
||||
"es",
|
||||
"it"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"design": [
|
||||
"voice_design"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "qwen3_tts_1_7b_base_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"Design",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/qwen3.md",
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_base_q8_0",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B Base Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-q8_0_v2.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_base_bf16",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B Base BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_base_orig",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B Base Original-Dtype GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_customvoice_q8_0",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B CustomVoice Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_customvoice_bf16",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B CustomVoice BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_voicedesign_q8_0",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B VoiceDesign Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF/qwen3-tts-12hz-1.7b-voicedesign-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_voicedesign_bf16",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B VoiceDesign BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF",
|
||||
"files": [
|
||||
"Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF/qwen3-tts-12hz-1.7b-voicedesign-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_1_7b_base_safetensors",
|
||||
"display_name": "Qwen3 TTS 12Hz 1.7B Base Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||
"files": [
|
||||
"config.json",
|
||||
"generation_config.json",
|
||||
"model.safetensors",
|
||||
"speech_tokenizer/config.json",
|
||||
"speech_tokenizer/model.safetensors",
|
||||
"tokenizer_config.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "qwen3_tts_0_6b_base_safetensors",
|
||||
"display_name": "Qwen3 TTS 12Hz 0.6B Base Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "Qwen3-TTS-12Hz-0.6B-Base",
|
||||
"files": [
|
||||
"config.json",
|
||||
"generation_config.json",
|
||||
"model.safetensors",
|
||||
"speech_tokenizer/config.json",
|
||||
"speech_tokenizer/model.safetensors",
|
||||
"tokenizer_config.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "Qwen/Qwen3-TTS-12Hz-0.6B-Base"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt",
|
||||
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"speech_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "speech_tokenizer_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt",
|
||||
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors",
|
||||
"speech_tokenizer_weights": "model:speech_tokenizer/model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,337 @@
|
||||
{
|
||||
"family": "seed_vc",
|
||||
"schema_version": 1,
|
||||
"display_name": "Seed-VC",
|
||||
"description": "Zero-shot voice conversion and singing voice conversion model for transferring timbre and style from reference audio, with low-latency realtime conversion and optional lightweight fine-tuning.",
|
||||
"category": "voice_conversion",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"vc",
|
||||
"svc"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"language_agnostic"
|
||||
],
|
||||
"capabilities": {
|
||||
"vc": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"svc": [
|
||||
"speaker_reference",
|
||||
"singing"
|
||||
]
|
||||
},
|
||||
"dependencies": [],
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "seed_vc_mlx_q8_0",
|
||||
"tags": [
|
||||
"VC",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/seed_vc.md",
|
||||
"docs/audio_tools.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"request": [
|
||||
{
|
||||
"name": "route",
|
||||
"type": "enum",
|
||||
"description": "Select the Seed-VC conversion route. Defaults to v2_vc for VC and v1_svc for SVC.",
|
||||
"values": [
|
||||
"v2_vc",
|
||||
"v1_svc",
|
||||
"v1_whisper_bigvgan_vc",
|
||||
"v1_xlsr_hift_vc"
|
||||
],
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "length_adjust",
|
||||
"type": "float",
|
||||
"description": "Output duration multiplier; must be positive, default 1.0.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 1.0
|
||||
},
|
||||
{
|
||||
"name": "num_inference_steps",
|
||||
"type": "int",
|
||||
"description": "Diffusion steps; default 30.",
|
||||
"required": false,
|
||||
"min": 1,
|
||||
"default": 30
|
||||
},
|
||||
{
|
||||
"name": "inference_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "V1 classifier-free guidance scale; default 0.7.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.7
|
||||
},
|
||||
{
|
||||
"name": "intelligibility_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "V2 classifier-free guidance scale for source-content intelligibility; default 0.7.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.7
|
||||
},
|
||||
{
|
||||
"name": "similarity_guidance_scale",
|
||||
"type": "float",
|
||||
"description": "V2 classifier-free guidance scale for target-speaker similarity; default 0.7.",
|
||||
"required": false,
|
||||
"min": 0.0,
|
||||
"default": 0.7
|
||||
},
|
||||
{
|
||||
"name": "voice_anonymization",
|
||||
"type": "bool",
|
||||
"description": "Use randomized average-voice conditioning instead of target-speaker conditioning for V2 anonymization; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "seed",
|
||||
"type": "int",
|
||||
"description": "Seed for V1/V2 diffusion noise and HiFT stochastic source excitation; omitted requests choose a random seed.",
|
||||
"required": false,
|
||||
"min": 0
|
||||
},
|
||||
{
|
||||
"name": "noise_path",
|
||||
"type": "path",
|
||||
"description": "Optional raw f32 noise file for deterministic V1/V2 diffusion noise and XLSR/HiFT source excitation.",
|
||||
"required": false
|
||||
},
|
||||
{
|
||||
"name": "f0_condition",
|
||||
"type": "bool",
|
||||
"description": "Enable V1 F0 conditioning for singing voice conversion; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "auto_f0_adjust",
|
||||
"type": "bool",
|
||||
"description": "Automatically adjust V1 source pitch toward the target pitch level; default false.",
|
||||
"required": false,
|
||||
"default": false
|
||||
},
|
||||
{
|
||||
"name": "semitone_shift",
|
||||
"type": "int",
|
||||
"description": "V1 pitch shift in semitones for singing voice conversion; default 0.",
|
||||
"required": false,
|
||||
"default": 0
|
||||
}
|
||||
],
|
||||
"session": [
|
||||
{
|
||||
"name": "weight_type",
|
||||
"type": "enum",
|
||||
"description": "Shared Seed-VC component weight storage type; default native, except RMVPE uses f32 unless overridden.",
|
||||
"preset": "weight_type_full",
|
||||
"required": false,
|
||||
"default": "native"
|
||||
}
|
||||
],
|
||||
"load": []
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "seed_vc_mlx_q8_0",
|
||||
"display_name": "SeedVC-MLX Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "SeedVC-MLX-GGUF",
|
||||
"files": [
|
||||
"SeedVC-MLX-GGUF/seed-vc-mlx-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "SeedVC-MLX-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "seed_vc_mlx_f16",
|
||||
"display_name": "SeedVC-MLX F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "SeedVC-MLX-GGUF",
|
||||
"files": [
|
||||
"SeedVC-MLX-GGUF/seed-vc-mlx-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "SeedVC-MLX-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "seed_vc_mlx_orig",
|
||||
"display_name": "SeedVC-MLX Original-Dtype GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "SeedVC-MLX-GGUF",
|
||||
"files": [
|
||||
"SeedVC-MLX-GGUF/seed-vc-mlx-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "SeedVC-MLX-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "seed_vc_mlx_safetensors",
|
||||
"display_name": "SeedVC-MLX Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "SeedVC-MLX",
|
||||
"files": [
|
||||
"seed_vc_manifest.json",
|
||||
"v2/ar.safetensors",
|
||||
"v2/cfm.safetensors"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "mlx-community/SeedVC-MLX"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"manifest": "model:seed_vc_manifest.json",
|
||||
"v2_wrapper_config": "model:v2/vc_wrapper.json",
|
||||
"astral_bsq32_config": "model:astral/bsq32.json",
|
||||
"astral_bsq2048_config": "model:astral/bsq2048.json",
|
||||
"v1_svc_config": "model:v1/svc.json",
|
||||
"v1_whisper_bigvgan_config": "model:v1/whisper_bigvgan.json",
|
||||
"v1_xlsr_hift_config": "model:v1/xlsr_hift.json",
|
||||
"hift_config": "model:hift/config.json",
|
||||
"bigvgan_22k_config": "model:bigvgan/v2_22khz_80band_256x/config.json",
|
||||
"bigvgan_44k_config": "model:bigvgan/v2_44khz_128band_512x/config.json",
|
||||
"whisper_small_config": "model:whisper-small/config.json",
|
||||
"hubert_large_config": "model:hubert-large-ll60k/config.json",
|
||||
"wav2vec2_xlsr_config": "model:wav2vec2-xls-r-300m/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"v2_ar_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "v2_ar_weights"
|
||||
},
|
||||
"v2_cfm_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "v2_cfm_weights"
|
||||
},
|
||||
"v1_svc_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "v1_svc_weights"
|
||||
},
|
||||
"v1_whisper_bigvgan_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "v1_whisper_bigvgan_weights"
|
||||
},
|
||||
"v1_xlsr_hift_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "v1_xlsr_hift_weights"
|
||||
},
|
||||
"astral_bsq32_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "astral_bsq32_weights"
|
||||
},
|
||||
"astral_bsq2048_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "astral_bsq2048_weights"
|
||||
},
|
||||
"campplus_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "campplus_weights"
|
||||
},
|
||||
"rmvpe_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "rmvpe_weights"
|
||||
},
|
||||
"hift_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "hift_weights"
|
||||
},
|
||||
"bigvgan_22k_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "bigvgan_22k_weights"
|
||||
},
|
||||
"bigvgan_44k_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "bigvgan_44k_weights"
|
||||
},
|
||||
"whisper_small_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "whisper_small_weights"
|
||||
},
|
||||
"hubert_large_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "hubert_large_weights"
|
||||
},
|
||||
"wav2vec2_xlsr_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "wav2vec2_xlsr_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"manifest": "model:seed_vc_manifest.json",
|
||||
"v2_wrapper_config": "model:v2/vc_wrapper.json",
|
||||
"astral_bsq32_config": "model:astral/bsq32.json",
|
||||
"astral_bsq2048_config": "model:astral/bsq2048.json",
|
||||
"v1_svc_config": "model:v1/svc.json",
|
||||
"v1_whisper_bigvgan_config": "model:v1/whisper_bigvgan.json",
|
||||
"v1_xlsr_hift_config": "model:v1/xlsr_hift.json",
|
||||
"hift_config": "model:hift/config.json",
|
||||
"bigvgan_22k_config": "model:bigvgan/v2_22khz_80band_256x/config.json",
|
||||
"bigvgan_44k_config": "model:bigvgan/v2_44khz_128band_512x/config.json",
|
||||
"whisper_small_config": "model:whisper-small/config.json",
|
||||
"hubert_large_config": "model:hubert-large-ll60k/config.json",
|
||||
"wav2vec2_xlsr_config": "model:wav2vec2-xls-r-300m/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"v2_ar_weights": "model:v2/ar.safetensors",
|
||||
"v2_cfm_weights": "model:v2/cfm.safetensors",
|
||||
"v1_svc_weights": "model:v1/svc.safetensors",
|
||||
"v1_whisper_bigvgan_weights": "model:v1/whisper_bigvgan.safetensors",
|
||||
"v1_xlsr_hift_weights": "model:v1/xlsr_hift.safetensors",
|
||||
"astral_bsq32_weights": "model:astral/bsq32.safetensors",
|
||||
"astral_bsq2048_weights": "model:astral/bsq2048.safetensors",
|
||||
"campplus_weights": "model:campplus/model.safetensors",
|
||||
"rmvpe_weights": "model:rmvpe/model.safetensors",
|
||||
"hift_weights": "model:hift/model.safetensors",
|
||||
"bigvgan_22k_weights": "model:bigvgan/v2_22khz_80band_256x/model.safetensors",
|
||||
"bigvgan_44k_weights": "model:bigvgan/v2_44khz_128band_512x/model.safetensors",
|
||||
"whisper_small_weights": "model:whisper-small/model.safetensors",
|
||||
"hubert_large_weights": "model:hubert-large-ll60k/model.safetensors",
|
||||
"wav2vec2_xlsr_weights": "model:wav2vec2-xls-r-300m/model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,183 @@
|
||||
{
|
||||
"family": "supertonic",
|
||||
"display_name": "Supertonic 3",
|
||||
"description": "Supertone on-device TTS model designed for fast local speech synthesis across 31 languages, with preset voices and compact deployment for browser, mobile, and desktop applications.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"en",
|
||||
"ko",
|
||||
"ja",
|
||||
"ar",
|
||||
"bg",
|
||||
"cs",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"es",
|
||||
"et",
|
||||
"fi",
|
||||
"fr",
|
||||
"hi",
|
||||
"hr",
|
||||
"hu",
|
||||
"id",
|
||||
"it",
|
||||
"lt",
|
||||
"lv",
|
||||
"nl",
|
||||
"pl",
|
||||
"pt",
|
||||
"ro",
|
||||
"ru",
|
||||
"sk",
|
||||
"sl",
|
||||
"sv",
|
||||
"tr",
|
||||
"uk",
|
||||
"vi"
|
||||
],
|
||||
"capabilities": {
|
||||
"tts": [
|
||||
"built_in_voices",
|
||||
"long_form"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "supertonic_3_orig",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "supertonic_3_q8_0",
|
||||
"display_name": "Supertonic 3 Q8_0 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Supertonic-3-GGUF",
|
||||
"files": [
|
||||
"Supertonic-3-GGUF/supertonic-3-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Supertonic-3-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "supertonic_3_f16",
|
||||
"display_name": "Supertonic 3 F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Supertonic-3-GGUF",
|
||||
"files": [
|
||||
"Supertonic-3-GGUF/supertonic-3-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Supertonic-3-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "supertonic_3_orig",
|
||||
"display_name": "Supertonic 3 Original-Dtype GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "Supertonic-3-GGUF",
|
||||
"files": [
|
||||
"Supertonic-3-GGUF/supertonic-3-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "Supertonic-3-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "supertonic_3_safetensors",
|
||||
"display_name": "Supertonic 3 Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "supertonic-3",
|
||||
"files": [
|
||||
"config/tts.json",
|
||||
"config/unicode_indexer.json",
|
||||
"ggml/supertonic.safetensors"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "mlx-community/supertonic-3-mlx"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"tts_config": "model:config/tts.json",
|
||||
"unicode_indexer": "model:config/unicode_indexer.json",
|
||||
"voice_style_F1": "model:voice_styles/F1.json",
|
||||
"voice_style_F2": "model:voice_styles/F2.json",
|
||||
"voice_style_F3": "model:voice_styles/F3.json",
|
||||
"voice_style_F4": "model:voice_styles/F4.json",
|
||||
"voice_style_F5": "model:voice_styles/F5.json",
|
||||
"voice_style_M1": "model:voice_styles/M1.json",
|
||||
"voice_style_M2": "model:voice_styles/M2.json",
|
||||
"voice_style_M3": "model:voice_styles/M3.json",
|
||||
"voice_style_M4": "model:voice_styles/M4.json",
|
||||
"voice_style_M5": "model:voice_styles/M5.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"tts_config": "model:config/tts.json",
|
||||
"unicode_indexer": "model:config/unicode_indexer.json",
|
||||
"voice_style_F1": "model:voice_styles/F1.json",
|
||||
"voice_style_F2": "model:voice_styles/F2.json",
|
||||
"voice_style_F3": "model:voice_styles/F3.json",
|
||||
"voice_style_F4": "model:voice_styles/F4.json",
|
||||
"voice_style_F5": "model:voice_styles/F5.json",
|
||||
"voice_style_M1": "model:voice_styles/M1.json",
|
||||
"voice_style_M2": "model:voice_styles/M2.json",
|
||||
"voice_style_M3": "model:voice_styles/M3.json",
|
||||
"voice_style_M4": "model:voice_styles/M4.json",
|
||||
"voice_style_M5": "model:voice_styles/M5.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:ggml/supertonic.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
{
|
||||
"family": "vevo2",
|
||||
"display_name": "Vevo2",
|
||||
"description": "Unified controllable framework for English and Chinese speech and singing voice generation, voice conversion, and editing, with tokenizers that disentangle content, prosody, melody, style, and timbre.",
|
||||
"category": "voice_conversion",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"music",
|
||||
"vc",
|
||||
"edit",
|
||||
"svc",
|
||||
"s2s"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en",
|
||||
"zh"
|
||||
],
|
||||
"capabilities": {
|
||||
"music": [
|
||||
"lyrics"
|
||||
],
|
||||
"vc": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"svc": [
|
||||
"speaker_reference",
|
||||
"singing"
|
||||
],
|
||||
"s2s": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"edit": [
|
||||
"prompt_editing"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "vevo2_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Music",
|
||||
"VC",
|
||||
"Edit",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/models/vevo2.md",
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "vevo2_q8_0",
|
||||
"display_name": "Vevo2 Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Vevo2-GGUF",
|
||||
"files": [
|
||||
"Vevo2-GGUF/vevo2-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Vevo2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "vevo2_f16",
|
||||
"display_name": "Vevo2 F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "Vevo2-GGUF",
|
||||
"files": [
|
||||
"Vevo2-GGUF/vevo2-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "Vevo2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "vevo2_orig",
|
||||
"display_name": "Vevo2 Original-Dtype GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "Vevo2-GGUF",
|
||||
"files": [
|
||||
"Vevo2-GGUF/vevo2-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "Vevo2-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"ar_config": "model:contentstyle_modeling/posttrained/config.json",
|
||||
"ar_amphion_config": "model:contentstyle_modeling/posttrained/amphion_config.json",
|
||||
"ar_generation_config": "model:contentstyle_modeling/posttrained/generation_config.json",
|
||||
"ar_tokenizer_config": "model:contentstyle_modeling/posttrained/tokenizer_config.json",
|
||||
"ar_tokenizer_json": "model:contentstyle_modeling/posttrained/tokenizer.json",
|
||||
"ar_vocab": "model:contentstyle_modeling/posttrained/vocab.json",
|
||||
"ar_merges": "model:contentstyle_modeling/posttrained/merges.txt",
|
||||
"ar_added_tokens": "model:contentstyle_modeling/posttrained/added_tokens.json",
|
||||
"ar_special_tokens": "model:contentstyle_modeling/posttrained/special_tokens_map.json",
|
||||
"fm_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/config.json",
|
||||
"fm_text_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/config.json",
|
||||
"vocoder_config": "model:vocoder/config.json",
|
||||
"whisper_config": "model:whisper-medium/config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"content_style_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "content_style_tokenizer_weights"
|
||||
},
|
||||
"prosody_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "prosody_tokenizer_weights"
|
||||
},
|
||||
"ar_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "ar_weights"
|
||||
},
|
||||
"fm_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "fm_weights"
|
||||
},
|
||||
"fm_whisper_stats": {
|
||||
"source": "weights:",
|
||||
"prefix": "fm_whisper_stats"
|
||||
},
|
||||
"fm_text_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "fm_text_weights"
|
||||
},
|
||||
"fm_text_whisper_stats": {
|
||||
"source": "weights:",
|
||||
"prefix": "fm_text_whisper_stats"
|
||||
},
|
||||
"vocoder_weights_0": {
|
||||
"source": "weights:",
|
||||
"prefix": "vocoder_weights_0"
|
||||
},
|
||||
"vocoder_weights_1": {
|
||||
"source": "weights:",
|
||||
"prefix": "vocoder_weights_1"
|
||||
},
|
||||
"vocoder_weights_2": {
|
||||
"source": "weights:",
|
||||
"prefix": "vocoder_weights_2"
|
||||
},
|
||||
"whisper_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "whisper_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"whisper": "../whisper-medium"
|
||||
},
|
||||
"files": {
|
||||
"ar_config": "model:contentstyle_modeling/posttrained/config.json",
|
||||
"ar_amphion_config": "model:contentstyle_modeling/posttrained/amphion_config.json",
|
||||
"ar_generation_config": "model:contentstyle_modeling/posttrained/generation_config.json",
|
||||
"ar_tokenizer_config": "model:contentstyle_modeling/posttrained/tokenizer_config.json",
|
||||
"ar_tokenizer_json": "model:contentstyle_modeling/posttrained/tokenizer.json",
|
||||
"ar_vocab": "model:contentstyle_modeling/posttrained/vocab.json",
|
||||
"ar_merges": "model:contentstyle_modeling/posttrained/merges.txt",
|
||||
"ar_added_tokens": "model:contentstyle_modeling/posttrained/added_tokens.json",
|
||||
"ar_special_tokens": "model:contentstyle_modeling/posttrained/special_tokens_map.json",
|
||||
"fm_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/config.json",
|
||||
"fm_text_config": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/config.json",
|
||||
"vocoder_config": "model:vocoder/config.json",
|
||||
"whisper_config": "whisper:config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"content_style_tokenizer_weights": "model:tokenizer/contentstyle_fvq16384_12.5hz/model.safetensors",
|
||||
"prosody_tokenizer_weights": "model:tokenizer/prosody_fvq512_6.25hz/model.safetensors",
|
||||
"ar_weights": "model:contentstyle_modeling/posttrained/model.safetensors",
|
||||
"fm_weights": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/model.safetensors",
|
||||
"fm_whisper_stats": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa/whisper_stats.safetensors",
|
||||
"fm_text_weights": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/model.safetensors",
|
||||
"fm_text_whisper_stats": "model:acoustic_modeling/fm_emilia101k_singnet7k_repa_text/whisper_stats.safetensors",
|
||||
"vocoder_weights_0": "model:vocoder/model.safetensors",
|
||||
"vocoder_weights_1": "model:vocoder/model_1.safetensors",
|
||||
"vocoder_weights_2": "model:vocoder/model_2.safetensors",
|
||||
"whisper_weights": "whisper:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
{
|
||||
"family": "vibevoice",
|
||||
"display_name": "VibeVoice",
|
||||
"description": "Microsoft long-form multi-speaker TTS model for expressive conversational audio such as podcasts, supporting up to 90 minutes of speech with as many as four speakers.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"en",
|
||||
"zh"
|
||||
],
|
||||
"capabilities": {
|
||||
"tts": [
|
||||
"multi_speaker",
|
||||
"long_form"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "vibevoice_1_5b_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "vibevoice_1_5b_q8_0",
|
||||
"display_name": "VibeVoice 1.5B Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "VibeVoice-1.5B-GGUF",
|
||||
"files": [
|
||||
"VibeVoice-1.5B-GGUF/vibevoice-1.5b-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "VibeVoice-1.5B-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "vibevoice_1_5b_bf16",
|
||||
"display_name": "VibeVoice 1.5B BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "VibeVoice-1.5B-GGUF",
|
||||
"files": [
|
||||
"VibeVoice-1.5B-GGUF/vibevoice-1.5b-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "VibeVoice-1.5B-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"preprocessor_config": "model:preprocessor_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_vocab": "model:vocab.json",
|
||||
"tokenizer_merges": "model:merges.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"preprocessor_config": "model:preprocessor_config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_vocab": "model:vocab.json",
|
||||
"tokenizer_merges": "model:merges.txt"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors.index.json"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
{
|
||||
"family": "vibevoice_asr",
|
||||
"display_name": "VibeVoice ASR",
|
||||
"description": "Microsoft long-form speech-to-text model that processes up to 60 minutes of audio in one pass and produces structured transcripts with speakers, timestamps, content, hotwords, and 50+ language support.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"auto",
|
||||
"51 languages"
|
||||
],
|
||||
"capabilities": {
|
||||
"asr": [
|
||||
"segments",
|
||||
"speaker_turns",
|
||||
"vad_chunking"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "vibevoice_asr_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "vibevoice_asr_q8_0",
|
||||
"display_name": "VibeVoice ASR Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "VibeVoice-ASR-GGUF",
|
||||
"files": [
|
||||
"VibeVoice-ASR-GGUF/vibevoice-asr-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "VibeVoice-ASR-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "vibevoice_asr_f16",
|
||||
"display_name": "VibeVoice ASR F16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "f16",
|
||||
"target_directory": "VibeVoice-ASR-GGUF",
|
||||
"files": [
|
||||
"VibeVoice-ASR-GGUF/vibevoice-asr-f16.gguf"
|
||||
],
|
||||
"strip_prefix": "VibeVoice-ASR-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_vocab": "model:vocab.json",
|
||||
"tokenizer_merges": "model:merges.txt"
|
||||
},
|
||||
"optional_files": {
|
||||
"preprocessor_config": "model:preprocessor_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"tokenizer_vocab": "model:vocab.json",
|
||||
"tokenizer_merges": "model:merges.txt"
|
||||
},
|
||||
"optional_files": {
|
||||
"preprocessor_config": "model:preprocessor_config.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors.index.json"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
{
|
||||
"family": "vietneu_tts",
|
||||
"display_name": "VieNeu-TTS v3 Turbo",
|
||||
"description": "On-device Vietnamese TTS model with instant voice cloning from 3-5 seconds of reference audio, English-Vietnamese code-switching, streaming playback, batched generation, and conversation mode.",
|
||||
"category": "tts",
|
||||
"status": "community",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone"
|
||||
],
|
||||
"modes": [
|
||||
"offline"
|
||||
],
|
||||
"languages": [
|
||||
"vi",
|
||||
"en"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "vietneu_tts_v3_turbo_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"GGUF"
|
||||
],
|
||||
"docs": [
|
||||
"docs/community_models/vietneu_tts.md",
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "vietneu_tts_v3_turbo_q8_0",
|
||||
"display_name": "VieNeu-TTS v3 Turbo GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "VieNeu-TTS-v3-Turbo-GGUF",
|
||||
"files": [
|
||||
"model.gguf"
|
||||
],
|
||||
"strip_prefix": ".",
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "phuocnguyen90/VieNeu-TTS-v3-Turbo-GGUF"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"generation_config": "model:generation_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "model_weights"
|
||||
},
|
||||
"speech_tokenizer_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "speech_tokenizer_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"speech_tokenizer_config": "model:speech_tokenizer/config.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"generation_config": "model:generation_config.json",
|
||||
"vocab": "model:vocab.json",
|
||||
"merges": "model:merges.txt",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json"
|
||||
},
|
||||
"tensors": {
|
||||
"model_weights": "model:model.safetensors",
|
||||
"speech_tokenizer_weights": "model:speech_tokenizer/model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
{
|
||||
"family": "voxcpm2",
|
||||
"display_name": "VoxCPM2",
|
||||
"description": "OpenBMB tokenizer-free TTS model supporting 30 languages and 9 Chinese dialects, with 48 kHz output, natural-language voice design, controllable short-reference voice cloning, and expressive style guidance.",
|
||||
"category": "tts",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"tts",
|
||||
"clone",
|
||||
"design"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"ar",
|
||||
"my",
|
||||
"zh",
|
||||
"zh dialects",
|
||||
"da",
|
||||
"nl",
|
||||
"en",
|
||||
"fi",
|
||||
"fr",
|
||||
"de",
|
||||
"el",
|
||||
"he",
|
||||
"hi",
|
||||
"id",
|
||||
"it",
|
||||
"ja",
|
||||
"km",
|
||||
"ko",
|
||||
"lo",
|
||||
"ms",
|
||||
"no",
|
||||
"pl",
|
||||
"pt",
|
||||
"ru",
|
||||
"es",
|
||||
"sw",
|
||||
"sv",
|
||||
"tl",
|
||||
"th",
|
||||
"tr",
|
||||
"vi"
|
||||
],
|
||||
"capabilities": {
|
||||
"clone": [
|
||||
"speaker_reference"
|
||||
],
|
||||
"design": [
|
||||
"voice_design"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "voxcpm2_q8_0",
|
||||
"tags": [
|
||||
"TTS",
|
||||
"Clone",
|
||||
"Design",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/tts.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "voxcpm2_q8_0",
|
||||
"display_name": "VoxCPM2 Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "VoxCPM2-GGUF",
|
||||
"files": [
|
||||
"VoxCPM2-GGUF/voxcpm2-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "VoxCPM2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "voxcpm2_bf16",
|
||||
"display_name": "VoxCPM2 BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "VoxCPM2-GGUF",
|
||||
"files": [
|
||||
"VoxCPM2-GGUF/voxcpm2-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "VoxCPM2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "voxcpm2_orig",
|
||||
"display_name": "VoxCPM2 Original-Dtype GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "orig",
|
||||
"target_directory": "VoxCPM2-GGUF",
|
||||
"files": [
|
||||
"VoxCPM2-GGUF/voxcpm2-orig.gguf"
|
||||
],
|
||||
"strip_prefix": "VoxCPM2-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "voxcpm2_safetensors",
|
||||
"display_name": "VoxCPM2 Safetensors",
|
||||
"format": "safetensors",
|
||||
"precision": "native",
|
||||
"target_directory": "VoxCPM2",
|
||||
"files": [
|
||||
"config.json",
|
||||
"model.safetensors",
|
||||
"tokenizer.json",
|
||||
"tokenizer_config.json"
|
||||
],
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "OpenBMB/VoxCPM2"
|
||||
}
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "weights"
|
||||
},
|
||||
"audiovae_weights": {
|
||||
"source": "weights:",
|
||||
"prefix": "audiovae_weights"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"tokenizer_config": "model:tokenizer_config.json",
|
||||
"tokenizer_json": "model:tokenizer.json",
|
||||
"special_tokens_map": "model:special_tokens_map.json"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors",
|
||||
"audiovae_weights": "model:audiovae.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
{
|
||||
"family": "voxtral_realtime",
|
||||
"display_name": "Voxtral Mini 4B Realtime",
|
||||
"description": "Mistral 13-language realtime ASR model with a natively streaming causal audio encoder, configurable low-latency transcription delay, and accuracy competitive with offline open-source systems.",
|
||||
"category": "asr",
|
||||
"status": "supported",
|
||||
"tasks": [
|
||||
"asr"
|
||||
],
|
||||
"modes": [
|
||||
"offline",
|
||||
"streaming"
|
||||
],
|
||||
"languages": [
|
||||
"en",
|
||||
"zh",
|
||||
"hi",
|
||||
"es",
|
||||
"ar",
|
||||
"fr",
|
||||
"pt",
|
||||
"ru",
|
||||
"de",
|
||||
"ja",
|
||||
"ko",
|
||||
"it",
|
||||
"nl"
|
||||
],
|
||||
"capabilities": {
|
||||
"asr": [
|
||||
"partial_results"
|
||||
]
|
||||
},
|
||||
"runtime": {
|
||||
"tags": [
|
||||
"gguf",
|
||||
"stream"
|
||||
]
|
||||
},
|
||||
"ui": {
|
||||
"recommended_package": "voxtral_realtime_q8_0",
|
||||
"tags": [
|
||||
"ASR",
|
||||
"GGUF",
|
||||
"Stream"
|
||||
],
|
||||
"docs": [
|
||||
"docs/asr.md",
|
||||
"docs/gguf.md"
|
||||
]
|
||||
},
|
||||
"package_defaults": {
|
||||
"download": {
|
||||
"kind": "huggingface_snapshot",
|
||||
"repo": "audio-cpp/audio.cpp-gguf",
|
||||
"revision": "main",
|
||||
"gated": false
|
||||
}
|
||||
},
|
||||
"packages": [
|
||||
{
|
||||
"id": "voxtral_realtime_q8_0",
|
||||
"display_name": "Voxtral Mini 4B Realtime Q8_0 GGUF",
|
||||
"default": true,
|
||||
"format": "gguf",
|
||||
"precision": "q8_0",
|
||||
"target_directory": "Voxtral-Mini-4B-Realtime-2602-GGUF",
|
||||
"files": [
|
||||
"Voxtral-Mini-4B-Realtime-2602-GGUF/voxtral-mini-4b-realtime-2602-q8_0.gguf"
|
||||
],
|
||||
"strip_prefix": "Voxtral-Mini-4B-Realtime-2602-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "voxtral_realtime_q4_k",
|
||||
"display_name": "Voxtral Mini 4B Realtime Q4_K GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "q4_k",
|
||||
"target_directory": "Voxtral-Mini-4B-Realtime-2602-GGUF",
|
||||
"files": [
|
||||
"Voxtral-Mini-4B-Realtime-2602-GGUF/voxtral-mini-4b-realtime-2602-q4_k.gguf"
|
||||
],
|
||||
"strip_prefix": "Voxtral-Mini-4B-Realtime-2602-GGUF"
|
||||
},
|
||||
{
|
||||
"id": "voxtral_realtime_bf16",
|
||||
"display_name": "Voxtral Mini 4B Realtime BF16 GGUF",
|
||||
"format": "gguf",
|
||||
"precision": "bf16",
|
||||
"target_directory": "Voxtral-Mini-4B-Realtime-2602-GGUF",
|
||||
"files": [
|
||||
"Voxtral-Mini-4B-Realtime-2602-GGUF/voxtral-mini-4b-realtime-2602-bf16.gguf"
|
||||
],
|
||||
"strip_prefix": "Voxtral-Mini-4B-Realtime-2602-GGUF"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"format": "gguf",
|
||||
"roots": {
|
||||
"model": ".",
|
||||
"weights": "$gguf"
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tekken": "model:tekken.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"params": "model:params.json",
|
||||
"readme": "model:README.md"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "weights:"
|
||||
}
|
||||
},
|
||||
{
|
||||
"format": "safetensors",
|
||||
"roots": {
|
||||
"model": "."
|
||||
},
|
||||
"files": {
|
||||
"config": "model:config.json",
|
||||
"generation_config": "model:generation_config.json",
|
||||
"processor_config": "model:processor_config.json",
|
||||
"tekken": "model:tekken.json"
|
||||
},
|
||||
"optional_files": {
|
||||
"params": "model:params.json",
|
||||
"readme": "model:README.md"
|
||||
},
|
||||
"tensors": {
|
||||
"weights": "model:model.safetensors"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,446 @@
|
||||
"""Owned audio.cpp server process launcher."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import socket
|
||||
import subprocess
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Mapping, Optional, Sequence
|
||||
|
||||
from .client import AudioCppClient, AudioCppClientError
|
||||
from .windows_job import WindowsKillOnCloseJob
|
||||
|
||||
|
||||
class AudioCppProcessError(RuntimeError):
|
||||
"""Base error for managed audio.cpp process failures."""
|
||||
|
||||
|
||||
class AudioCppProcessStartupError(AudioCppProcessError):
|
||||
"""The owned audio.cpp server failed before becoming ready."""
|
||||
|
||||
|
||||
def find_free_loopback_port() -> int:
|
||||
"""Ask the OS for a currently unused IPv4 loopback port."""
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
|
||||
sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
sock.bind(("127.0.0.1", 0))
|
||||
return int(sock.getsockname()[1])
|
||||
|
||||
|
||||
def normalize_audio_cpp_task(task: Any) -> str:
|
||||
normalized = str(task or "tts").strip().lower().replace("-", "_").replace(" ", "_")
|
||||
aliases = {
|
||||
"clone": "clon",
|
||||
"cloning": "clon",
|
||||
"voice_clone": "clon",
|
||||
"voice_cloning": "clon",
|
||||
"design": "vdes",
|
||||
"voice_design": "vdes",
|
||||
"voice_designer": "vdes",
|
||||
}
|
||||
return aliases.get(normalized, normalized)
|
||||
|
||||
|
||||
def _first(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
|
||||
for key in keys:
|
||||
if key in config and config[key] is not None:
|
||||
return config[key]
|
||||
return default
|
||||
|
||||
|
||||
def _resolve_existing_path(value: Any, label: str, *, executable: bool = False) -> Path:
|
||||
raw = os.path.expandvars(os.path.expanduser(str(value or "").strip()))
|
||||
if not raw:
|
||||
raise AudioCppProcessStartupError(f"Missing audio.cpp {label}")
|
||||
candidate = Path(raw)
|
||||
if executable and not candidate.exists():
|
||||
located = shutil.which(raw)
|
||||
if located:
|
||||
candidate = Path(located)
|
||||
candidate = candidate.resolve()
|
||||
if not candidate.exists():
|
||||
raise AudioCppProcessStartupError(f"audio.cpp {label} does not exist: {candidate}")
|
||||
if executable and not candidate.is_file():
|
||||
raise AudioCppProcessStartupError(f"audio.cpp {label} is not a file: {candidate}")
|
||||
return candidate
|
||||
|
||||
|
||||
def _safe_model_id(value: Any, family: str) -> str:
|
||||
raw = str(value or family or "audio-cpp-model").strip()
|
||||
cleaned = "".join(char if char.isalnum() or char in "._-" else "-" for char in raw)
|
||||
cleaned = cleaned.strip("-.")
|
||||
return cleaned or "audio-cpp-model"
|
||||
|
||||
|
||||
def _option_dict(config: Mapping[str, Any], key: str) -> Dict[str, Any]:
|
||||
value = config.get(key, {})
|
||||
if value is None:
|
||||
return {}
|
||||
if not isinstance(value, Mapping):
|
||||
raise AudioCppProcessStartupError(f"audio.cpp {key} must be a JSON object")
|
||||
return dict(value)
|
||||
|
||||
|
||||
class AudioCppServerProcess:
|
||||
"""One suite-owned, single-model ``audiocpp_server`` process."""
|
||||
|
||||
def __init__(self, config: Mapping[str, Any]) -> None:
|
||||
self.config = dict(config)
|
||||
self.binary_path = _resolve_existing_path(
|
||||
_first(
|
||||
self.config,
|
||||
"binary_path",
|
||||
"server_binary",
|
||||
"executable_path",
|
||||
"audio_cpp_binary",
|
||||
),
|
||||
"server executable",
|
||||
executable=True,
|
||||
)
|
||||
self.model_path = _resolve_existing_path(
|
||||
_first(self.config, "model_path", "package_path", "gguf_path"),
|
||||
"model path",
|
||||
)
|
||||
self.family = str(_first(self.config, "family", "model_family", default="")).strip()
|
||||
if not self.family:
|
||||
raise AudioCppProcessStartupError("Missing audio.cpp model family")
|
||||
self.task = normalize_audio_cpp_task(_first(self.config, "task", default="tts"))
|
||||
self.model_id = _safe_model_id(
|
||||
_first(self.config, "model_id", "server_model_id", "package_id"),
|
||||
self.family,
|
||||
)
|
||||
self.backend = str(_first(self.config, "backend", default="cuda")).strip().lower()
|
||||
if self.backend == "auto":
|
||||
self.backend = "cuda"
|
||||
if self.backend not in {"cuda", "hip", "cpu", "vulkan", "metal"}:
|
||||
raise AudioCppProcessStartupError(f"Unsupported audio.cpp backend: {self.backend}")
|
||||
self.device = self._resolve_device_index(
|
||||
_first(self.config, "device_index", "device", default=0)
|
||||
)
|
||||
# Match audio.cpp's CLI default instead of the server's conservative
|
||||
# one-thread example configuration.
|
||||
self.threads = max(1, int(_first(self.config, "threads", default=4)))
|
||||
self.port = int(_first(self.config, "port", "server_port", default=0) or 0)
|
||||
self.startup_timeout = max(
|
||||
0.1, float(_first(self.config, "startup_timeout", "startup_timeout_seconds", default=30.0))
|
||||
)
|
||||
self.request_timeout = max(
|
||||
0.1, float(_first(self.config, "request_timeout", "request_timeout_seconds", default=600.0))
|
||||
)
|
||||
self.connect_timeout = max(
|
||||
0.05, float(_first(self.config, "connect_timeout", "connect_timeout_seconds", default=2.0))
|
||||
)
|
||||
self.stop_timeout = max(
|
||||
0.1, float(_first(self.config, "stop_timeout", "stop_timeout_seconds", default=5.0))
|
||||
)
|
||||
self._lock = threading.RLock()
|
||||
self._process: Optional[subprocess.Popen] = None
|
||||
self._client: Optional[AudioCppClient] = None
|
||||
self._temp_dir: Optional[tempfile.TemporaryDirectory] = None
|
||||
self._config_path: Optional[Path] = None
|
||||
self._log_path: Optional[Path] = None
|
||||
self._log_handle = None
|
||||
self._closed = False
|
||||
self._parent_job: Optional[WindowsKillOnCloseJob] = None
|
||||
|
||||
@staticmethod
|
||||
def _resolve_device_index(value: Any) -> int:
|
||||
text = str(value).strip().lower()
|
||||
if text in {"auto", "cuda", "hip", "cpu", "vulkan", "metal", ""}:
|
||||
return 0
|
||||
if ":" in text:
|
||||
text = text.rsplit(":", 1)[-1]
|
||||
try:
|
||||
return max(0, int(text))
|
||||
except ValueError as exc:
|
||||
raise AudioCppProcessStartupError(
|
||||
f"audio.cpp device must be an integer index, got {value!r}"
|
||||
) from exc
|
||||
|
||||
@property
|
||||
def process(self) -> Optional[subprocess.Popen]:
|
||||
return self._process
|
||||
|
||||
@property
|
||||
def client(self) -> AudioCppClient:
|
||||
if self._client is None:
|
||||
raise AudioCppProcessError("audio.cpp server process has not started")
|
||||
return self._client
|
||||
|
||||
@property
|
||||
def config_path(self) -> Optional[Path]:
|
||||
return self._config_path
|
||||
|
||||
@property
|
||||
def log_path(self) -> Optional[Path]:
|
||||
return self._log_path
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
if self.port <= 0:
|
||||
raise AudioCppProcessError("audio.cpp server port is not allocated")
|
||||
return f"http://127.0.0.1:{self.port}"
|
||||
|
||||
@property
|
||||
def running(self) -> bool:
|
||||
return self._process is not None and self._process.poll() is None
|
||||
|
||||
def _model_spec_override(self) -> Optional[Path]:
|
||||
if "model_spec_override" in self.config:
|
||||
explicit = self.config.get("model_spec_override")
|
||||
if explicit in (None, "", False):
|
||||
return None
|
||||
path = _resolve_existing_path(explicit, "model spec override")
|
||||
else:
|
||||
path = (Path(__file__).resolve().parent / "model_specs").resolve()
|
||||
if not path.is_dir():
|
||||
raise AudioCppProcessStartupError(
|
||||
"Bundled audio.cpp release-0.5.1 model specs are missing: " + str(path)
|
||||
)
|
||||
return path
|
||||
|
||||
def _build_server_config(self) -> Dict[str, Any]:
|
||||
model: Dict[str, Any] = {
|
||||
"id": self.model_id,
|
||||
"family": self.family,
|
||||
"path": str(self.model_path),
|
||||
"task": self.task,
|
||||
"mode": str(_first(self.config, "mode", "run_mode", default="offline")),
|
||||
"lazy": bool(_first(self.config, "lazy", "lazy_load", default=True)),
|
||||
"load_options": _option_dict(self.config, "load_options"),
|
||||
"session_options": _option_dict(self.config, "session_options"),
|
||||
"default_request_options": _option_dict(self.config, "default_request_options"),
|
||||
}
|
||||
for source_key, target_key in (
|
||||
("config_id", "config"),
|
||||
("weight_id", "weight"),
|
||||
("voice_presets", "voice_presets"),
|
||||
("default_voice_preset", "default_voice_preset"),
|
||||
("model_busy_timeout_ms", "busy_timeout_ms"),
|
||||
):
|
||||
if source_key in self.config and self.config[source_key] is not None:
|
||||
model[target_key] = self.config[source_key]
|
||||
|
||||
server: Dict[str, Any] = {
|
||||
"host": "127.0.0.1",
|
||||
"port": self.port,
|
||||
"cors_origins": "",
|
||||
"backend": self.backend,
|
||||
"device": self.device,
|
||||
"threads": self.threads,
|
||||
"lazy_load": bool(_first(self.config, "lazy_load", default=True)),
|
||||
"log_request_body": False,
|
||||
"max_request_body_bytes": int(
|
||||
_first(self.config, "max_request_body_bytes", default=2 * 1024 * 1024 * 1024)
|
||||
),
|
||||
"busy_timeout_ms": int(_first(self.config, "busy_timeout_ms", default=300000)),
|
||||
"models": [model],
|
||||
}
|
||||
model_spec_override = self._model_spec_override()
|
||||
if model_spec_override is not None:
|
||||
server["model_spec_override"] = str(model_spec_override)
|
||||
return server
|
||||
|
||||
def _prepare_files(self) -> None:
|
||||
temp_root = _first(self.config, "temp_root", "runtime_temp_root")
|
||||
if temp_root:
|
||||
Path(str(temp_root)).expanduser().resolve().mkdir(parents=True, exist_ok=True)
|
||||
self._temp_dir = tempfile.TemporaryDirectory(
|
||||
prefix="tts_audio_cpp_",
|
||||
dir=str(Path(str(temp_root)).expanduser().resolve()) if temp_root else None,
|
||||
)
|
||||
temp_path = Path(self._temp_dir.name)
|
||||
self._config_path = temp_path / "server.json"
|
||||
log_dir = _first(self.config, "log_dir")
|
||||
if log_dir:
|
||||
resolved_log_dir = Path(str(log_dir)).expanduser().resolve()
|
||||
resolved_log_dir.mkdir(parents=True, exist_ok=True)
|
||||
self._log_path = resolved_log_dir / f"audio_cpp_{self.model_id}_{self.port}.log"
|
||||
else:
|
||||
self._log_path = temp_path / "server.log"
|
||||
with self._config_path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
json.dump(self._build_server_config(), handle, ensure_ascii=False, indent=2)
|
||||
handle.write("\n")
|
||||
self._log_handle = self._log_path.open("ab", buffering=0)
|
||||
|
||||
def _command(self) -> list[str]:
|
||||
binary_args = _first(self.config, "binary_args", "launcher_args", default=[])
|
||||
if binary_args is None:
|
||||
binary_args = []
|
||||
if isinstance(binary_args, (str, bytes)) or not isinstance(binary_args, Sequence):
|
||||
raise AudioCppProcessStartupError("audio.cpp binary_args must be a list")
|
||||
return [
|
||||
str(self.binary_path),
|
||||
*[str(item) for item in binary_args],
|
||||
"--config",
|
||||
str(self._config_path),
|
||||
]
|
||||
|
||||
def start(self) -> "AudioCppServerProcess":
|
||||
with self._lock:
|
||||
if self._closed:
|
||||
raise AudioCppProcessError("audio.cpp server process launcher is closed")
|
||||
if self.running:
|
||||
return self
|
||||
if self.port <= 0:
|
||||
self.port = find_free_loopback_port()
|
||||
try:
|
||||
self._prepare_files()
|
||||
env = os.environ.copy()
|
||||
extra_env = _first(self.config, "environment", "env", default={})
|
||||
if extra_env:
|
||||
if not isinstance(extra_env, Mapping):
|
||||
raise AudioCppProcessStartupError(
|
||||
"audio.cpp environment must be an object"
|
||||
)
|
||||
env.update({str(key): str(value) for key, value in extra_env.items()})
|
||||
env.setdefault("PYTHONUTF8", "1")
|
||||
visible_console = bool(self.config.get("show_server_console", False))
|
||||
creationflags = 0
|
||||
if os.name == "nt":
|
||||
creationflags = getattr(
|
||||
subprocess,
|
||||
"CREATE_NEW_CONSOLE" if visible_console else "CREATE_NO_WINDOW",
|
||||
0,
|
||||
)
|
||||
self._process = subprocess.Popen(
|
||||
self._command(),
|
||||
cwd=str(self.binary_path.parent),
|
||||
env=env,
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=None if visible_console else self._log_handle,
|
||||
stderr=None if visible_console else subprocess.STDOUT,
|
||||
shell=False,
|
||||
creationflags=creationflags,
|
||||
)
|
||||
if os.name == "nt":
|
||||
self._parent_job = WindowsKillOnCloseJob()
|
||||
self._parent_job.assign(int(self._process._handle))
|
||||
self._client = AudioCppClient(
|
||||
self.base_url,
|
||||
connect_timeout=self.connect_timeout,
|
||||
request_timeout=self.request_timeout,
|
||||
)
|
||||
self._wait_until_ready()
|
||||
return self
|
||||
except Exception as exc:
|
||||
log_tail = self.read_log_tail()
|
||||
self._terminate_exact_process()
|
||||
self._client = None
|
||||
self._cleanup_files()
|
||||
if isinstance(exc, AudioCppProcessStartupError):
|
||||
raise
|
||||
suffix = f"\nServer log tail:\n{log_tail}" if log_tail else ""
|
||||
raise AudioCppProcessStartupError(
|
||||
f"Failed to start audio.cpp server: {exc}{suffix}"
|
||||
) from exc
|
||||
|
||||
def _wait_until_ready(self) -> None:
|
||||
deadline = time.monotonic() + self.startup_timeout
|
||||
last_error: Optional[BaseException] = None
|
||||
while time.monotonic() < deadline:
|
||||
if self._process is None or self._process.poll() is not None:
|
||||
code = self._process.poll() if self._process is not None else "unknown"
|
||||
log_tail = self.read_log_tail()
|
||||
suffix = f"\nServer log tail:\n{log_tail}" if log_tail else ""
|
||||
raise AudioCppProcessStartupError(
|
||||
f"audio.cpp server exited during startup with code {code}{suffix}"
|
||||
)
|
||||
try:
|
||||
health = self.client.health(timeout=min(self.connect_timeout, 0.5))
|
||||
status = str(health.get("status", "")).lower()
|
||||
if status in {"ok", "ready", "healthy"} or health:
|
||||
models = self.client.models(timeout=min(self.connect_timeout, 1.0))
|
||||
if any(str(item.get("id")) == self.model_id for item in models):
|
||||
return
|
||||
last_error = AudioCppProcessStartupError(
|
||||
f"audio.cpp server did not register expected model id '{self.model_id}'"
|
||||
)
|
||||
except AudioCppClientError as exc:
|
||||
last_error = exc
|
||||
time.sleep(0.05)
|
||||
log_tail = self.read_log_tail()
|
||||
suffix = f"\nServer log tail:\n{log_tail}" if log_tail else ""
|
||||
raise AudioCppProcessStartupError(
|
||||
f"audio.cpp server was not ready after {self.startup_timeout:.1f}s"
|
||||
+ (f": {last_error}" if last_error else "")
|
||||
+ suffix
|
||||
)
|
||||
|
||||
def read_log_tail(self, max_bytes: int = 16384) -> str:
|
||||
path = self._log_path
|
||||
if path is None or not path.exists():
|
||||
return ""
|
||||
try:
|
||||
if self._log_handle is not None:
|
||||
self._log_handle.flush()
|
||||
with path.open("rb") as handle:
|
||||
size = path.stat().st_size
|
||||
handle.seek(max(0, size - max(1, int(max_bytes))))
|
||||
return handle.read().decode("utf-8", errors="replace").strip()
|
||||
except OSError:
|
||||
return ""
|
||||
|
||||
def _terminate_exact_process(self) -> None:
|
||||
process = self._process
|
||||
if process is None:
|
||||
return
|
||||
try:
|
||||
if process.poll() is None:
|
||||
process.terminate()
|
||||
try:
|
||||
process.wait(timeout=self.stop_timeout)
|
||||
except subprocess.TimeoutExpired:
|
||||
process.kill()
|
||||
process.wait(timeout=self.stop_timeout)
|
||||
finally:
|
||||
self._process = None
|
||||
if self._parent_job is not None:
|
||||
self._parent_job.close()
|
||||
self._parent_job = None
|
||||
|
||||
def _cleanup_files(self) -> None:
|
||||
if self._log_handle is not None:
|
||||
try:
|
||||
self._log_handle.close()
|
||||
except OSError:
|
||||
pass
|
||||
self._log_handle = None
|
||||
if self._temp_dir is not None:
|
||||
try:
|
||||
self._temp_dir.cleanup()
|
||||
except OSError:
|
||||
pass
|
||||
self._temp_dir = None
|
||||
|
||||
def close(self) -> None:
|
||||
with self._lock:
|
||||
if self._closed:
|
||||
return
|
||||
self._closed = True
|
||||
self._terminate_exact_process()
|
||||
self._client = None
|
||||
self._cleanup_files()
|
||||
|
||||
stop = close
|
||||
|
||||
def __enter__(self) -> "AudioCppServerProcess":
|
||||
return self.start()
|
||||
|
||||
def __exit__(self, exc_type, exc, traceback) -> None:
|
||||
self.close()
|
||||
|
||||
|
||||
__all__ = [
|
||||
"AudioCppProcessError",
|
||||
"AudioCppProcessStartupError",
|
||||
"AudioCppServerProcess",
|
||||
"find_free_loopback_port",
|
||||
"normalize_audio_cpp_task",
|
||||
]
|
||||
@@ -0,0 +1,609 @@
|
||||
"""Resolve workflow and machine-local audio.cpp configuration into one session config."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import urllib.parse
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Mapping, Optional, Sequence
|
||||
|
||||
from .settings import AudioCppSettings, load_settings
|
||||
|
||||
|
||||
class AudioCppResolutionError(RuntimeError):
|
||||
"""Raised when an audio.cpp engine configuration cannot be made runnable."""
|
||||
|
||||
|
||||
class _SuiteDownloadProgress:
|
||||
"""Match UnifiedDownloader's single-line console progress convention."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._completed: set[str] = set()
|
||||
|
||||
def __call__(self, label: str, downloaded: int, total: Optional[int]) -> None:
|
||||
if not total or total <= 0:
|
||||
return
|
||||
filename = Path(label).name
|
||||
percent = min(100.0, downloaded * 100.0 / total)
|
||||
print(f"\r📥 Downloading {filename}: {percent:.1f}%", end="", flush=True)
|
||||
if downloaded >= total and label not in self._completed:
|
||||
self._completed.add(label)
|
||||
print()
|
||||
|
||||
|
||||
def _print_download_block(
|
||||
title: str,
|
||||
*,
|
||||
model: str,
|
||||
description: str,
|
||||
repository: str,
|
||||
target: Path,
|
||||
size_bytes: Optional[int] = None,
|
||||
) -> None:
|
||||
"""Use the same boxed pre-download summary as the Suite engine downloaders."""
|
||||
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"📦 {title}")
|
||||
print("=" * 60)
|
||||
print(f"Model: {model}")
|
||||
print(f"Description: {description}")
|
||||
print(f"Repository: {repository}")
|
||||
print(f"Download size: {_format_download_size(size_bytes)}")
|
||||
print(f"Target: {target}")
|
||||
print(f"{'=' * 60}\n")
|
||||
|
||||
|
||||
def _format_download_size(size_bytes: Optional[int]) -> str:
|
||||
if size_bytes is None or size_bytes < 0:
|
||||
return "Unknown"
|
||||
if size_bytes >= 1024**3:
|
||||
return f"{size_bytes / 1024**3:.2f} GB"
|
||||
return f"{size_bytes / 1024**2:.1f} MB"
|
||||
|
||||
|
||||
_EXTERNAL_MODES = {
|
||||
"external",
|
||||
"external_server",
|
||||
"existing_server",
|
||||
"remote_server",
|
||||
"server",
|
||||
"connect",
|
||||
}
|
||||
_OWNED_MODES = {
|
||||
"owned",
|
||||
"owned_process",
|
||||
"existing_binary",
|
||||
"managed_binary",
|
||||
"managed",
|
||||
"binary",
|
||||
"local_binary",
|
||||
}
|
||||
_TASK_ALIASES = {
|
||||
"clone": "clon",
|
||||
"cloning": "clon",
|
||||
"voice_clone": "clon",
|
||||
"voice_cloning": "clon",
|
||||
"voice_design": "vdes",
|
||||
"design": "vdes",
|
||||
"voice_conversion": "vc",
|
||||
"speech_to_speech": "s2s",
|
||||
"singing_voice_conversion": "svc",
|
||||
}
|
||||
|
||||
|
||||
def _catalog_module():
|
||||
from . import catalog
|
||||
|
||||
return catalog
|
||||
|
||||
|
||||
def _discovery_module():
|
||||
from . import discovery
|
||||
|
||||
return discovery
|
||||
|
||||
|
||||
def _downloader_module():
|
||||
from . import downloader
|
||||
|
||||
return downloader
|
||||
|
||||
|
||||
def _runtime_installer_module():
|
||||
from . import runtime_installer
|
||||
|
||||
return runtime_installer
|
||||
|
||||
|
||||
def _flatten_config(config: Mapping[str, Any]) -> dict[str, Any]:
|
||||
if not isinstance(config, Mapping):
|
||||
raise TypeError("audio.cpp config must be a mapping")
|
||||
nested = config.get("config")
|
||||
flattened = dict(nested) if isinstance(nested, Mapping) else {}
|
||||
flattened.update({key: value for key, value in config.items() if key != "config"})
|
||||
return flattened
|
||||
|
||||
|
||||
def _has_value(value: Any) -> bool:
|
||||
if value is None:
|
||||
return False
|
||||
if isinstance(value, str):
|
||||
return bool(value.strip())
|
||||
if isinstance(value, (list, tuple, set, dict)):
|
||||
return bool(value)
|
||||
return True
|
||||
|
||||
|
||||
def _first_value(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
|
||||
for key in keys:
|
||||
if key in config and _has_value(config[key]):
|
||||
return config[key]
|
||||
return default
|
||||
|
||||
|
||||
def _merge_settings(config: Mapping[str, Any], settings: AudioCppSettings) -> dict[str, Any]:
|
||||
"""Fill blank/absent workflow fields without replacing explicit values."""
|
||||
|
||||
merged = settings.to_mapping()
|
||||
for key, value in config.items():
|
||||
if _has_value(value) or key not in merged:
|
||||
merged[key] = value
|
||||
return merged
|
||||
|
||||
|
||||
def _normalize_mode(value: Any) -> str:
|
||||
normalized = re.sub(r"[^a-z0-9]+", "_", str(value or "auto").strip().lower()).strip("_")
|
||||
if normalized in {"", "auto"}:
|
||||
return "auto"
|
||||
if normalized in _EXTERNAL_MODES:
|
||||
return "external_server"
|
||||
if normalized in _OWNED_MODES:
|
||||
return "owned_process"
|
||||
raise AudioCppResolutionError(f"Unsupported audio.cpp connection mode: {value!r}")
|
||||
|
||||
|
||||
def _canonical_url(value: Any) -> str:
|
||||
raw = str(value or "").strip()
|
||||
parsed = urllib.parse.urlsplit(raw)
|
||||
if parsed.scheme not in {"http", "https"} or not parsed.netloc:
|
||||
raise AudioCppResolutionError(
|
||||
"audio.cpp external mode requires a valid HTTP(S) server_url; "
|
||||
f"received {value!r}"
|
||||
)
|
||||
if parsed.query or parsed.fragment:
|
||||
raise AudioCppResolutionError(
|
||||
"audio.cpp external server_url must not contain a query string or fragment"
|
||||
)
|
||||
return urllib.parse.urlunsplit(
|
||||
(parsed.scheme.lower(), parsed.netloc, parsed.path.rstrip("/"), "", "")
|
||||
)
|
||||
|
||||
|
||||
def _path_text(value: Any) -> str:
|
||||
return os.path.expandvars(os.path.expanduser(os.fspath(value))).strip()
|
||||
|
||||
|
||||
def _existing_path(value: Any, label: str, *, file_only: bool = False) -> Path:
|
||||
raw = _path_text(value)
|
||||
if not raw:
|
||||
raise AudioCppResolutionError(f"Missing audio.cpp {label}")
|
||||
candidate = Path(raw)
|
||||
if file_only and not candidate.exists():
|
||||
located = shutil.which(raw)
|
||||
if located:
|
||||
candidate = Path(located)
|
||||
candidate = candidate.resolve()
|
||||
if not candidate.exists():
|
||||
raise AudioCppResolutionError(f"audio.cpp {label} does not exist: {candidate}")
|
||||
if file_only and not candidate.is_file():
|
||||
raise AudioCppResolutionError(f"audio.cpp {label} is not a file: {candidate}")
|
||||
return candidate
|
||||
|
||||
|
||||
def _canonical_model_id(value: Any, fallback: str) -> str:
|
||||
raw = str(value or fallback or "audio-cpp-model").strip()
|
||||
cleaned = "".join(char if char.isalnum() or char in "._-" else "-" for char in raw)
|
||||
return cleaned.strip("-.") or "audio-cpp-model"
|
||||
|
||||
|
||||
def _device_index(value: Any) -> int:
|
||||
text = str(value if value is not None else 0).strip().lower()
|
||||
if text in {"", "auto", "cuda", "cpu", "vulkan", "metal", "hip"}:
|
||||
return 0
|
||||
if ":" in text:
|
||||
text = text.rsplit(":", 1)[-1]
|
||||
try:
|
||||
index = int(text)
|
||||
except ValueError as exc:
|
||||
raise AudioCppResolutionError(
|
||||
f"audio.cpp device must be a non-negative integer index, got {value!r}"
|
||||
) from exc
|
||||
if index < 0:
|
||||
raise AudioCppResolutionError(
|
||||
f"audio.cpp device must be a non-negative integer index, got {value!r}"
|
||||
)
|
||||
return index
|
||||
|
||||
|
||||
def _as_bool(value: Any, default: bool = False) -> bool:
|
||||
if value is None:
|
||||
return default
|
||||
if isinstance(value, str):
|
||||
normalized = value.strip().lower()
|
||||
if normalized in {"1", "true", "yes", "on"}:
|
||||
return True
|
||||
if normalized in {"0", "false", "no", "off", ""}:
|
||||
return False
|
||||
raise AudioCppResolutionError(f"Invalid boolean value in audio.cpp config: {value!r}")
|
||||
return bool(value)
|
||||
|
||||
|
||||
def _cuda_available() -> bool:
|
||||
try:
|
||||
import torch
|
||||
|
||||
return bool(torch.cuda.is_available())
|
||||
except (ImportError, RuntimeError):
|
||||
return False
|
||||
|
||||
|
||||
def _resolve_backend(config: Mapping[str, Any], settings: AudioCppSettings, *, owned: bool) -> str:
|
||||
requested = str(_first_value(config, "backend", default="auto") or "auto").strip().lower()
|
||||
if owned and requested == "auto" and settings.runtime_backend in {"cpu", "cuda"}:
|
||||
requested = settings.runtime_backend
|
||||
if requested == "auto":
|
||||
return "cuda" if owned and _cuda_available() else ("cpu" if owned else "auto")
|
||||
supported = {"cuda", "cpu", "vulkan", "metal", "hip"} if owned else {
|
||||
"cuda",
|
||||
"cpu",
|
||||
"vulkan",
|
||||
"metal",
|
||||
"hip",
|
||||
}
|
||||
if requested not in supported:
|
||||
raise AudioCppResolutionError(f"Unsupported audio.cpp backend: {requested!r}")
|
||||
return requested
|
||||
|
||||
|
||||
def _explicit_roots(config: Mapping[str, Any]) -> Optional[Sequence[Any]]:
|
||||
value = _first_value(config, "model_roots", "model_search_roots")
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, (str, os.PathLike)):
|
||||
return [value]
|
||||
if isinstance(value, Sequence):
|
||||
return list(value)
|
||||
raise AudioCppResolutionError("audio.cpp model_roots must be a path or list of paths")
|
||||
|
||||
|
||||
def _installed_runtime_binary(runtime_root: Any, backend: str, runtime_api) -> Optional[Path]:
|
||||
if not _has_value(runtime_root):
|
||||
return None
|
||||
root = Path(_path_text(runtime_root)).resolve()
|
||||
direct = root / "audiocpp_server.exe"
|
||||
if direct.is_file() and direct.stat().st_size > 0:
|
||||
return direct.resolve()
|
||||
if backend not in {"cpu", "cuda"}:
|
||||
return None
|
||||
candidate = runtime_api.runtime_install_path(root, backend) / "audiocpp_server.exe"
|
||||
if candidate.is_file() and candidate.stat().st_size > 0:
|
||||
return candidate.resolve()
|
||||
return None
|
||||
|
||||
|
||||
def _external_task(config: Mapping[str, Any]) -> str:
|
||||
requested = str(
|
||||
_first_value(config, "requested_task", "task", default="auto") or "auto"
|
||||
).strip().lower().replace("-", "_").replace(" ", "_")
|
||||
requested = _TASK_ALIASES.get(requested, requested)
|
||||
supported = {"auto", "tts", "clon", "vdes", "vc", "s2s", "svc", "asr", "diar"}
|
||||
if requested not in supported:
|
||||
raise AudioCppResolutionError(f"Unsupported audio.cpp task: {requested!r}")
|
||||
return requested
|
||||
|
||||
|
||||
def resolve_audio_cpp_config(
|
||||
config: Mapping[str, Any],
|
||||
*,
|
||||
external: Optional[bool] = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Return a complete canonical config without starting an audio.cpp session.
|
||||
|
||||
External mode intentionally returns before importing model discovery or either
|
||||
installer. Owned mode may install only when the corresponding workflow flag
|
||||
explicitly permits it, and all installs target suite-managed storage.
|
||||
"""
|
||||
|
||||
workflow = _flatten_config(config)
|
||||
settings = load_settings()
|
||||
merged = _merge_settings(workflow, settings)
|
||||
workflow_mode = _normalize_mode(_first_value(workflow, "connection_mode", "source", default="auto"))
|
||||
workflow_url = _first_value(
|
||||
workflow, "server_url", "endpoint", "base_url", "external_server_url"
|
||||
)
|
||||
merged_url = _first_value(
|
||||
merged, "server_url", "endpoint", "base_url", "external_server_url"
|
||||
)
|
||||
workflow_binary = _first_value(
|
||||
workflow, "binary_path", "server_binary", "executable_path", "audio_cpp_binary"
|
||||
)
|
||||
|
||||
if external is True:
|
||||
mode = "external_server"
|
||||
elif external is False:
|
||||
mode = "owned_process"
|
||||
elif workflow_mode != "auto":
|
||||
mode = workflow_mode
|
||||
elif _has_value(workflow_url):
|
||||
# A URL deliberately supplied by the workflow outranks machine defaults.
|
||||
mode = "external_server"
|
||||
elif _has_value(workflow_binary):
|
||||
mode = "owned_process"
|
||||
elif settings.connection_mode == "external" and _has_value(merged_url):
|
||||
mode = "external_server"
|
||||
elif _has_value(settings.executable_path):
|
||||
mode = "owned_process"
|
||||
else:
|
||||
mode = "owned_process"
|
||||
|
||||
device_index = _device_index(_first_value(merged, "device_index", "device", default=0))
|
||||
family = str(_first_value(merged, "family", "model_family", default="") or "").strip()
|
||||
package_value = str(_first_value(merged, "package_id", default="auto") or "auto").strip()
|
||||
|
||||
if mode == "external_server":
|
||||
endpoint = _canonical_url(merged_url)
|
||||
package_id = package_value or "auto"
|
||||
result = dict(merged)
|
||||
explicit_model_id = _first_value(workflow, "model_id", "server_model_id")
|
||||
result.update(
|
||||
{
|
||||
"connection_mode": "external_server",
|
||||
"server_url": endpoint,
|
||||
"external_server_url": endpoint,
|
||||
"family": family,
|
||||
"package_id": package_id,
|
||||
"task": _external_task(workflow),
|
||||
"backend": _resolve_backend(workflow, settings, owned=False),
|
||||
"device": device_index,
|
||||
"device_index": device_index,
|
||||
"binary_path": "",
|
||||
"model_path": "",
|
||||
}
|
||||
)
|
||||
if explicit_model_id:
|
||||
result["model_id"] = _canonical_model_id(explicit_model_id, "")
|
||||
else:
|
||||
# Omission is meaningful: the external session will select the
|
||||
# server's sole /v1/models entry. Its lookup treats blank as an ID.
|
||||
result.pop("model_id", None)
|
||||
result.pop("server_model_id", None)
|
||||
return result
|
||||
|
||||
if not family or family.lower() == "auto":
|
||||
raise AudioCppResolutionError(
|
||||
"Owned audio.cpp mode requires a model family from the pinned release-0.5.1 catalog"
|
||||
)
|
||||
|
||||
catalog_api = _catalog_module()
|
||||
try:
|
||||
catalog = catalog_api.load_catalog()
|
||||
family_record = catalog.family(family)
|
||||
package_id = (
|
||||
family_record.recommended_package_id
|
||||
if package_value.lower() in {"", "auto"}
|
||||
else package_value
|
||||
)
|
||||
package = catalog.package(package_id)
|
||||
if package.family != family:
|
||||
raise AudioCppResolutionError(
|
||||
f"audio.cpp package {package_id!r} belongs to {package.family!r}, not {family!r}"
|
||||
)
|
||||
requested_task = _first_value(
|
||||
workflow, "requested_task", "task", default=_first_value(merged, "task", default="auto")
|
||||
)
|
||||
task = catalog_api.resolve_task(family, package_id, requested=str(requested_task or "auto"))
|
||||
except AudioCppResolutionError:
|
||||
raise
|
||||
except (KeyError, TypeError, ValueError) as exc:
|
||||
raise AudioCppResolutionError(f"Invalid audio.cpp model selection: {exc}") from exc
|
||||
|
||||
backend = _resolve_backend(workflow, settings, owned=True)
|
||||
discovery_api = _discovery_module()
|
||||
managed_model_root = discovery_api.default_managed_model_root(settings=settings)
|
||||
|
||||
explicit_model = _first_value(workflow, "model_path", "package_path", "gguf_path")
|
||||
if _has_value(explicit_model):
|
||||
model_path = _existing_path(explicit_model, "model path")
|
||||
else:
|
||||
resolved_model = discovery_api.resolve_model(
|
||||
package,
|
||||
_explicit_roots(workflow),
|
||||
catalog=catalog,
|
||||
settings=settings,
|
||||
)
|
||||
if resolved_model is not None:
|
||||
model_path = Path(resolved_model.path).resolve()
|
||||
elif _as_bool(_first_value(workflow, "auto_download_model", default=False)):
|
||||
downloader_api = _downloader_module()
|
||||
size_resolver = getattr(downloader_api, "package_download_size", None)
|
||||
download_size = (
|
||||
size_resolver(package, catalog=catalog) if callable(size_resolver) else None
|
||||
)
|
||||
_print_download_block(
|
||||
"audio.cpp Model Download",
|
||||
model=package.display_name,
|
||||
description=(
|
||||
f"{family_record.display_name} {package.precision.upper()} "
|
||||
f"{package.format.upper()} package"
|
||||
),
|
||||
repository=package.repo,
|
||||
target=(Path(managed_model_root) / package.target_directory).resolve(),
|
||||
size_bytes=download_size,
|
||||
)
|
||||
print(f"📥 Downloading {package_id} directly (no cache)")
|
||||
try:
|
||||
download_result = downloader_api.install_package(
|
||||
package,
|
||||
managed_model_root,
|
||||
catalog=catalog,
|
||||
progress=_SuiteDownloadProgress(),
|
||||
)
|
||||
except Exception as exc:
|
||||
raise AudioCppResolutionError(
|
||||
f"Failed to install audio.cpp package {package_id!r} in managed storage "
|
||||
f"{managed_model_root}: {exc}"
|
||||
) from exc
|
||||
model_path = Path(download_result.path).resolve()
|
||||
print(f"✅ Downloaded: {model_path}")
|
||||
else:
|
||||
searched = discovery_api.resolve_model_roots(
|
||||
_explicit_roots(workflow), settings=settings
|
||||
)
|
||||
raise AudioCppResolutionError(
|
||||
f"audio.cpp package {package_id!r} is not installed. Searched: "
|
||||
+ ", ".join(str(Path(root)) for root in searched)
|
||||
+ ". Provide model_path or enable auto_download_model."
|
||||
)
|
||||
|
||||
dependency_session_options: Dict[str, Any] = {}
|
||||
try:
|
||||
from .capabilities import get_package_dependencies
|
||||
|
||||
dependencies = get_package_dependencies(package_id)
|
||||
except (ImportError, KeyError, TypeError, ValueError) as exc:
|
||||
raise AudioCppResolutionError(
|
||||
f"Cannot resolve audio.cpp dependencies for {package_id!r}: {exc}"
|
||||
) from exc
|
||||
dependency_roots = discovery_api.resolve_model_roots(
|
||||
_explicit_roots(workflow), settings=settings
|
||||
)
|
||||
for dependency in dependencies:
|
||||
dependency_package = dependency["package"]
|
||||
dependency_path = discovery_api.find_installed_package(
|
||||
dependency_package,
|
||||
dependency_roots,
|
||||
settings=settings,
|
||||
)
|
||||
if dependency_path is None:
|
||||
if not _as_bool(_first_value(workflow, "auto_download_model", default=False)):
|
||||
raise AudioCppResolutionError(
|
||||
f"audio.cpp package {package_id!r} requires {dependency_package.id!r}, "
|
||||
"which is not installed. Enable auto_download_model to install it."
|
||||
)
|
||||
downloader_api = _downloader_module()
|
||||
_print_download_block(
|
||||
"audio.cpp Dependency Download",
|
||||
model=dependency_package.display_name,
|
||||
description=f"Required by {family_record.display_name}",
|
||||
repository=dependency_package.repo,
|
||||
target=(
|
||||
Path(managed_model_root) / dependency_package.target_directory
|
||||
).resolve(),
|
||||
size_bytes=int(dependency["estimated_download_bytes"]),
|
||||
)
|
||||
print(f"📥 Downloading {dependency_package.id} directly (no cache)")
|
||||
try:
|
||||
dependency_result = downloader_api.install_package(
|
||||
dependency_package,
|
||||
managed_model_root,
|
||||
progress=_SuiteDownloadProgress(),
|
||||
)
|
||||
except Exception as exc:
|
||||
raise AudioCppResolutionError(
|
||||
f"Failed to install dependency {dependency_package.id!r} for "
|
||||
f"audio.cpp package {package_id!r}: {exc}"
|
||||
) from exc
|
||||
dependency_path = Path(dependency_result.path).resolve()
|
||||
print(f"✅ Downloaded dependency: {dependency_path}")
|
||||
dependency_session_options[str(dependency["session_option"])] = str(
|
||||
Path(dependency_path).resolve()
|
||||
)
|
||||
|
||||
explicit_binary = _first_value(
|
||||
workflow, "binary_path", "server_binary", "executable_path", "audio_cpp_binary"
|
||||
)
|
||||
managed_runtime_root = Path(managed_model_root).expanduser().resolve().parent / "runtime"
|
||||
if _has_value(explicit_binary):
|
||||
binary_path = _existing_path(explicit_binary, "server executable", file_only=True)
|
||||
elif _has_value(settings.executable_path):
|
||||
binary_path = _existing_path(
|
||||
settings.executable_path, "configured server executable", file_only=True
|
||||
)
|
||||
else:
|
||||
runtime_api = _runtime_installer_module()
|
||||
binary_path = _installed_runtime_binary(settings.runtime_root, backend, runtime_api)
|
||||
if binary_path is None:
|
||||
binary_path = _installed_runtime_binary(managed_runtime_root, backend, runtime_api)
|
||||
if binary_path is None and backend == "cpu":
|
||||
# The official CUDA profile also contains the CPU backend. Reuse it
|
||||
# before downloading a second executable profile solely for CPU mode.
|
||||
binary_path = _installed_runtime_binary(settings.runtime_root, "cuda", runtime_api)
|
||||
if binary_path is None and backend == "cpu":
|
||||
binary_path = _installed_runtime_binary(managed_runtime_root, "cuda", runtime_api)
|
||||
if binary_path is None:
|
||||
if backend not in {"cpu", "cuda"}:
|
||||
raise AudioCppResolutionError(
|
||||
f"The pinned managed audio.cpp runtime has no {backend!r} Windows artifact. "
|
||||
"Provide binary_path for this backend."
|
||||
)
|
||||
if not _as_bool(_first_value(workflow, "auto_download_runtime", default=False)):
|
||||
expected = runtime_api.runtime_install_path(managed_runtime_root, backend)
|
||||
raise AudioCppResolutionError(
|
||||
f"audio.cpp {backend} server runtime is not installed at {expected}. "
|
||||
"Provide binary_path or enable auto_download_runtime."
|
||||
)
|
||||
runtime_manifest = runtime_api.get_runtime_manifest(backend)
|
||||
_print_download_block(
|
||||
"audio.cpp Runtime Download",
|
||||
model=f"audio.cpp {runtime_manifest.release_version} ({backend})",
|
||||
description=(
|
||||
f"Official Windows {runtime_manifest.profile} runtime, "
|
||||
f"pinned to {runtime_manifest.release_tag}"
|
||||
),
|
||||
repository="0xShug0/audio.cpp",
|
||||
target=runtime_api.runtime_install_path(managed_runtime_root, backend),
|
||||
size_bytes=sum(asset.size for asset in runtime_manifest.assets),
|
||||
)
|
||||
print(f"📥 Downloading audio.cpp {backend} runtime directly (no cache)")
|
||||
try:
|
||||
runtime_result = runtime_api.install_windows_runtime(
|
||||
managed_runtime_root,
|
||||
backend,
|
||||
progress=_SuiteDownloadProgress(),
|
||||
)
|
||||
except Exception as exc:
|
||||
raise AudioCppResolutionError(
|
||||
f"Failed to install the pinned audio.cpp {backend} balance runtime in "
|
||||
f"{managed_runtime_root}: {exc}"
|
||||
) from exc
|
||||
binary_path = Path(runtime_result.executable).resolve()
|
||||
print(f"✅ Downloaded: {binary_path}")
|
||||
|
||||
result = dict(merged)
|
||||
session_options = dict(result.get("session_options") or {})
|
||||
session_options.update(dependency_session_options)
|
||||
result.update(
|
||||
{
|
||||
"connection_mode": "owned_process",
|
||||
"server_url": "",
|
||||
"external_server_url": "",
|
||||
"family": family,
|
||||
"package_id": package_id,
|
||||
"model_id": _canonical_model_id(
|
||||
_first_value(workflow, "model_id", "server_model_id"), package_id
|
||||
),
|
||||
"task": task,
|
||||
"backend": backend,
|
||||
"device": device_index,
|
||||
"device_index": device_index,
|
||||
"binary_path": str(binary_path),
|
||||
"model_path": str(model_path),
|
||||
"session_options": session_options,
|
||||
}
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
__all__ = ["AudioCppResolutionError", "resolve_audio_cpp_config"]
|
||||
@@ -0,0 +1,252 @@
|
||||
"""Verified Windows audio.cpp release-0.5.1 runtime installer."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
import shutil
|
||||
import stat
|
||||
import sys
|
||||
import tempfile
|
||||
import uuid
|
||||
import zipfile
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path, PurePosixPath
|
||||
from typing import Callable, Mapping, Optional, Sequence, Union
|
||||
|
||||
from .catalog import AUDIO_CPP_RELEASE_COMMIT, AUDIO_CPP_RELEASE_TAG, AUDIO_CPP_RELEASE_VERSION
|
||||
from .downloader import AudioCppDownloadError, ProgressCallback, download_url_to_path
|
||||
|
||||
|
||||
class RuntimeInstallError(RuntimeError):
|
||||
"""Raised when a managed audio.cpp runtime cannot be verified or installed."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RuntimeAsset:
|
||||
filename: str
|
||||
url: str
|
||||
size: int
|
||||
sha256: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RuntimeManifest:
|
||||
backend: str
|
||||
assets: tuple[RuntimeAsset, ...]
|
||||
required_files: tuple[str, ...]
|
||||
release_version: str = AUDIO_CPP_RELEASE_VERSION
|
||||
release_tag: str = AUDIO_CPP_RELEASE_TAG
|
||||
release_commit: str = AUDIO_CPP_RELEASE_COMMIT
|
||||
profile: str = "balance"
|
||||
platform: str = "windows"
|
||||
|
||||
|
||||
_RELEASE_URL = "https://github.com/0xShug0/audio.cpp/releases/download/release-0.5.1"
|
||||
|
||||
WINDOWS_RUNTIME_MANIFESTS: Mapping[str, RuntimeManifest] = {
|
||||
"cpu": RuntimeManifest(
|
||||
backend="cpu",
|
||||
assets=(
|
||||
RuntimeAsset(
|
||||
filename="audiocpp-windows-cpu-balance-238ab6a9.zip",
|
||||
url=f"{_RELEASE_URL}/audiocpp-windows-cpu-balance-238ab6a9.zip",
|
||||
size=11_435_334,
|
||||
sha256="c9db54d75becfc9dfa6930d469b6f201f0f3919e36dfa7f488d0ee754ab5fe23",
|
||||
),
|
||||
),
|
||||
required_files=("audiocpp_server.exe", "audiocpp_cli.exe"),
|
||||
),
|
||||
"cuda": RuntimeManifest(
|
||||
backend="cuda",
|
||||
assets=(
|
||||
RuntimeAsset(
|
||||
filename="audiocpp-windows-cuda-balance-238ab6a9.zip",
|
||||
url=f"{_RELEASE_URL}/audiocpp-windows-cuda-balance-238ab6a9.zip",
|
||||
size=248_519_503,
|
||||
sha256="7e20f1fa984960b327700365a9a8434dc8effa0fcaa8d871c6cf5a35bf77b2a7",
|
||||
),
|
||||
RuntimeAsset(
|
||||
filename="audiocpp-windows-cuda-runtime.zip",
|
||||
url=f"{_RELEASE_URL}/audiocpp-windows-cuda-runtime.zip",
|
||||
size=575_505_446,
|
||||
sha256="46016655aff8f050806d81efd0fe256c15b86527935bfb3896208d4cac6b5ff8",
|
||||
),
|
||||
),
|
||||
required_files=(
|
||||
"audiocpp_server.exe",
|
||||
"audiocpp_cli.exe",
|
||||
"cublas64_13.dll",
|
||||
"cublasLt64_13.dll",
|
||||
"cufft64_12.dll",
|
||||
),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RuntimeInstallResult:
|
||||
manifest: RuntimeManifest
|
||||
path: Path
|
||||
executable: Path
|
||||
already_present: bool = False
|
||||
|
||||
|
||||
def get_runtime_manifest(backend: str) -> RuntimeManifest:
|
||||
normalized = str(backend).strip().lower()
|
||||
try:
|
||||
return WINDOWS_RUNTIME_MANIFESTS[normalized]
|
||||
except KeyError as exc:
|
||||
raise RuntimeInstallError("audio.cpp runtime backend must be 'cpu' or 'cuda'") from exc
|
||||
|
||||
|
||||
def runtime_install_path(runtime_root: Union[str, Path], backend: str) -> Path:
|
||||
manifest = get_runtime_manifest(backend)
|
||||
return (
|
||||
Path(runtime_root).expanduser()
|
||||
/ f"release-{manifest.release_version}"
|
||||
/ f"windows-{manifest.backend}-{manifest.profile}"
|
||||
)
|
||||
|
||||
|
||||
def runtime_is_complete(path: Union[str, Path], manifest: RuntimeManifest) -> bool:
|
||||
root = Path(path)
|
||||
return root.is_dir() and all(
|
||||
(root / relative).is_file() and (root / relative).stat().st_size > 0
|
||||
for relative in manifest.required_files
|
||||
)
|
||||
|
||||
|
||||
def _sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as handle:
|
||||
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _safe_member_path(member_name: str) -> Path:
|
||||
normalized = member_name.replace("\\", "/")
|
||||
path = PurePosixPath(normalized)
|
||||
if (
|
||||
not normalized
|
||||
or path.is_absolute()
|
||||
or ".." in path.parts
|
||||
or (path.parts and ":" in path.parts[0])
|
||||
):
|
||||
raise RuntimeInstallError(f"Unsafe path in runtime archive: {member_name!r}")
|
||||
return Path(*path.parts)
|
||||
|
||||
|
||||
def _extract_zip(archive: Path, destination: Path) -> None:
|
||||
try:
|
||||
with zipfile.ZipFile(archive) as bundle:
|
||||
for info in bundle.infolist():
|
||||
relative = _safe_member_path(info.filename)
|
||||
unix_mode = info.external_attr >> 16
|
||||
if stat.S_ISLNK(unix_mode):
|
||||
raise RuntimeInstallError(f"Runtime archive contains a symlink: {info.filename}")
|
||||
target = destination / relative
|
||||
if info.is_dir():
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
continue
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
with bundle.open(info, "r") as source, target.open("wb") as output:
|
||||
shutil.copyfileobj(source, output, length=1024 * 1024)
|
||||
except (OSError, zipfile.BadZipFile) as exc:
|
||||
raise RuntimeInstallError(f"Cannot extract runtime archive {archive.name}: {exc}") from exc
|
||||
|
||||
|
||||
def _remove_path(path: Path) -> None:
|
||||
if path.is_symlink() or path.is_file():
|
||||
path.unlink(missing_ok=True)
|
||||
elif path.is_dir():
|
||||
shutil.rmtree(path)
|
||||
|
||||
|
||||
def _publish_directory(staging: Path, target: Path, overwrite: bool) -> None:
|
||||
if not target.exists() and not target.is_symlink():
|
||||
staging.rename(target)
|
||||
return
|
||||
if not overwrite:
|
||||
raise RuntimeInstallError(f"Runtime target exists but is incomplete: {target}")
|
||||
backup = target.with_name(f".{target.name}.{uuid.uuid4().hex}.backup")
|
||||
target.rename(backup)
|
||||
try:
|
||||
staging.rename(target)
|
||||
except BaseException:
|
||||
if not target.exists() and backup.exists():
|
||||
backup.rename(target)
|
||||
raise
|
||||
_remove_path(backup)
|
||||
|
||||
|
||||
def install_windows_runtime(
|
||||
runtime_root: Union[str, Path],
|
||||
backend: str,
|
||||
*,
|
||||
overwrite: bool = False,
|
||||
manifest: Optional[RuntimeManifest] = None,
|
||||
platform_name: Optional[str] = None,
|
||||
timeout: int = 600,
|
||||
opener=None,
|
||||
progress: Optional[ProgressCallback] = None,
|
||||
) -> RuntimeInstallResult:
|
||||
"""Download, hash, extract, validate, and atomically publish a runtime."""
|
||||
|
||||
platform_value = (platform_name or sys.platform).lower()
|
||||
if platform_value not in {"win32", "windows"}:
|
||||
raise RuntimeInstallError("Managed audio.cpp release binaries are currently Windows-only")
|
||||
selected = manifest or get_runtime_manifest(backend)
|
||||
if selected.backend != str(backend).strip().lower():
|
||||
raise RuntimeInstallError("Runtime manifest backend does not match the requested backend")
|
||||
target = runtime_install_path(runtime_root, selected.backend)
|
||||
if runtime_is_complete(target, selected) and not overwrite:
|
||||
return RuntimeInstallResult(
|
||||
selected, target, target / "audiocpp_server.exe", already_present=True
|
||||
)
|
||||
if (target.exists() or target.is_symlink()) and not overwrite:
|
||||
raise RuntimeInstallError(f"Runtime target exists but is incomplete: {target}")
|
||||
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
work = Path(tempfile.mkdtemp(prefix=f".{target.name}.", suffix=".staging", dir=target.parent))
|
||||
archives = work / "archives"
|
||||
payload = work / "payload"
|
||||
archives.mkdir()
|
||||
payload.mkdir()
|
||||
try:
|
||||
for asset in selected.assets:
|
||||
archive = archives / asset.filename
|
||||
try:
|
||||
downloaded = download_url_to_path(
|
||||
asset.url,
|
||||
archive,
|
||||
timeout=timeout,
|
||||
opener=opener,
|
||||
progress=progress,
|
||||
progress_label=asset.filename,
|
||||
)
|
||||
except AudioCppDownloadError as exc:
|
||||
raise RuntimeInstallError(str(exc)) from exc
|
||||
if downloaded != asset.size:
|
||||
raise RuntimeInstallError(
|
||||
f"Runtime asset {asset.filename} has size {downloaded}; expected {asset.size}"
|
||||
)
|
||||
actual_hash = _sha256(archive)
|
||||
if actual_hash.lower() != asset.sha256.lower():
|
||||
raise RuntimeInstallError(
|
||||
f"SHA256 mismatch for {asset.filename}: expected {asset.sha256}, got {actual_hash}"
|
||||
)
|
||||
_extract_zip(archive, payload)
|
||||
|
||||
if not runtime_is_complete(payload, selected):
|
||||
missing = [
|
||||
item
|
||||
for item in selected.required_files
|
||||
if not (payload / item).is_file() or (payload / item).stat().st_size == 0
|
||||
]
|
||||
raise RuntimeInstallError(f"Runtime staging validation failed; missing: {missing}")
|
||||
_publish_directory(payload, target, overwrite=overwrite)
|
||||
return RuntimeInstallResult(selected, target, target / "audiocpp_server.exe")
|
||||
finally:
|
||||
_remove_path(work)
|
||||
@@ -0,0 +1,626 @@
|
||||
"""Keyed audio.cpp sessions and ComfyUI lifecycle integration."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import atexit
|
||||
import ipaddress
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
import urllib.parse
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Mapping, Optional
|
||||
|
||||
from .client import (
|
||||
AudioCppClient,
|
||||
AudioCppConnectionError,
|
||||
AudioCppTaskResult,
|
||||
AudioCppTimeoutError,
|
||||
)
|
||||
from .lifecycle import AudioCppRuntimeProxy
|
||||
from .process import AudioCppServerProcess, normalize_audio_cpp_task
|
||||
from .resolver import resolve_audio_cpp_config
|
||||
|
||||
|
||||
def _warn(message: str, exc: Optional[BaseException] = None) -> None:
|
||||
"""Emit diagnostics without assuming a UTF-8 Windows console."""
|
||||
text = f"WARNING: {message}"
|
||||
if exc is not None:
|
||||
text += f": {exc}"
|
||||
encoding = getattr(sys.stderr, "encoding", None) or "ascii"
|
||||
try:
|
||||
text = text.encode(encoding, errors="replace").decode(encoding, errors="replace")
|
||||
except LookupError:
|
||||
text = text.encode("ascii", errors="replace").decode("ascii")
|
||||
print(text, file=sys.stderr)
|
||||
|
||||
|
||||
def _flatten_config(config: Mapping[str, Any]) -> Dict[str, Any]:
|
||||
if not isinstance(config, Mapping):
|
||||
raise TypeError("audio.cpp config must be a mapping")
|
||||
nested = config.get("config")
|
||||
flattened: Dict[str, Any] = dict(nested) if isinstance(nested, Mapping) else {}
|
||||
flattened.update({key: value for key, value in config.items() if key != "config"})
|
||||
return flattened
|
||||
|
||||
|
||||
def _first(config: Mapping[str, Any], *keys: str, default: Any = None) -> Any:
|
||||
for key in keys:
|
||||
if key in config and config[key] is not None:
|
||||
return config[key]
|
||||
return default
|
||||
|
||||
|
||||
def _connection_mode(config: Mapping[str, Any]) -> str:
|
||||
raw = str(_first(config, "connection_mode", "source", default="auto"))
|
||||
normalized = re.sub(r"[^a-z0-9]+", "_", raw.strip().lower()).strip("_")
|
||||
external_aliases = {
|
||||
"existing_server",
|
||||
"external_server",
|
||||
"server",
|
||||
"remote_server",
|
||||
"connect",
|
||||
}
|
||||
owned_aliases = {
|
||||
"owned_process",
|
||||
"existing_binary",
|
||||
"managed_binary",
|
||||
"managed",
|
||||
"binary",
|
||||
"local_binary",
|
||||
}
|
||||
if normalized in external_aliases:
|
||||
return "external_server"
|
||||
if normalized in owned_aliases:
|
||||
return "owned_process"
|
||||
if normalized not in {"", "auto"}:
|
||||
raise ValueError(f"Unsupported audio.cpp connection mode: {raw}")
|
||||
endpoint = _first(config, "server_url", "endpoint", "base_url")
|
||||
return "external_server" if endpoint else "owned_process"
|
||||
|
||||
|
||||
def _is_loopback_url(url: str) -> bool:
|
||||
parsed = urllib.parse.urlsplit(url)
|
||||
host = (parsed.hostname or "").strip().lower()
|
||||
if host == "localhost":
|
||||
return True
|
||||
try:
|
||||
return ipaddress.ip_address(host).is_loopback
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
def _canonical_url(url: Any) -> str:
|
||||
parsed = urllib.parse.urlsplit(str(url or "").strip())
|
||||
if parsed.scheme not in {"http", "https"} or not parsed.netloc:
|
||||
raise ValueError(f"Invalid audio.cpp external server URL: {url!r}")
|
||||
if parsed.query or parsed.fragment:
|
||||
raise ValueError("audio.cpp external server URL must not contain a query or fragment")
|
||||
return urllib.parse.urlunsplit(
|
||||
(parsed.scheme, parsed.netloc, parsed.path.rstrip("/"), "", "")
|
||||
)
|
||||
|
||||
|
||||
def _safe_model_id(config: Mapping[str, Any]) -> str:
|
||||
raw = _first(
|
||||
config,
|
||||
"model_id",
|
||||
"server_model_id",
|
||||
"package_id",
|
||||
default=_first(config, "family", "model_family", default="audio-cpp-model"),
|
||||
)
|
||||
value = str(raw or "audio-cpp-model").strip()
|
||||
cleaned = "".join(char if char.isalnum() or char in "._-" else "-" for char in value)
|
||||
return cleaned.strip("-.") or "audio-cpp-model"
|
||||
|
||||
|
||||
def _jsonable(value: Any) -> Any:
|
||||
if isinstance(value, Mapping):
|
||||
return {str(key): _jsonable(item) for key, item in sorted(value.items(), key=lambda item: str(item[0]))}
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [_jsonable(item) for item in value]
|
||||
if isinstance(value, Path):
|
||||
return str(value.expanduser().resolve())
|
||||
if isinstance(value, (str, int, float, bool)) or value is None:
|
||||
return value
|
||||
return repr(value)
|
||||
|
||||
|
||||
def _session_key(config: Mapping[str, Any], mode: str, model_id: str, endpoint: str = "") -> str:
|
||||
if mode == "external_server":
|
||||
identity = {
|
||||
"mode": mode,
|
||||
"endpoint": endpoint,
|
||||
"model_id": model_id,
|
||||
}
|
||||
else:
|
||||
identity_keys = (
|
||||
"binary_path",
|
||||
"server_binary",
|
||||
"executable_path",
|
||||
"audio_cpp_binary",
|
||||
"binary_args",
|
||||
"model_path",
|
||||
"package_path",
|
||||
"gguf_path",
|
||||
"family",
|
||||
"model_family",
|
||||
"task",
|
||||
"backend",
|
||||
"device",
|
||||
"device_index",
|
||||
"threads",
|
||||
"load_options",
|
||||
"session_options",
|
||||
"default_request_options",
|
||||
"config_id",
|
||||
"weight_id",
|
||||
"model_spec_override",
|
||||
"show_server_console",
|
||||
)
|
||||
identity = {"mode": mode, "model_id": model_id}
|
||||
for key in identity_keys:
|
||||
if key in config:
|
||||
identity[key] = config[key]
|
||||
return json.dumps(_jsonable(identity), ensure_ascii=True, sort_keys=True, separators=(",", ":"))
|
||||
|
||||
|
||||
def _normalize_request_paths(request: Mapping[str, Any]) -> Dict[str, Any]:
|
||||
normalized = dict(request)
|
||||
path_fields = {
|
||||
"voice_ref",
|
||||
"audio_path",
|
||||
"source_audio",
|
||||
"target_voice",
|
||||
"prosody_ref",
|
||||
"style_ref",
|
||||
}
|
||||
for key in path_fields:
|
||||
value = normalized.get(key)
|
||||
if isinstance(value, os.PathLike) or (isinstance(value, str) and value.strip()):
|
||||
normalized[key] = str(
|
||||
Path(os.path.expandvars(os.path.expanduser(str(value)))).resolve()
|
||||
)
|
||||
return normalized
|
||||
|
||||
|
||||
class AudioCppSession:
|
||||
"""Persistent external connection or restartable suite-owned audio.cpp server."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
config: Mapping[str, Any],
|
||||
*,
|
||||
owned: bool,
|
||||
model_id: str,
|
||||
client: Optional[AudioCppClient] = None,
|
||||
endpoint: str = "",
|
||||
model_metadata: Optional[Mapping[str, Any]] = None,
|
||||
) -> None:
|
||||
self.config = dict(config)
|
||||
self.owned = bool(owned)
|
||||
self.model_id = str(model_id)
|
||||
self.endpoint = endpoint
|
||||
self.model_metadata = dict(model_metadata or {})
|
||||
self.family = str(
|
||||
_first(
|
||||
self.model_metadata,
|
||||
"family",
|
||||
default=_first(self.config, "family", "model_family", default=""),
|
||||
)
|
||||
or ""
|
||||
)
|
||||
self.task = normalize_audio_cpp_task(
|
||||
_first(
|
||||
self.model_metadata,
|
||||
"task",
|
||||
default=_first(self.config, "task", default="tts"),
|
||||
)
|
||||
)
|
||||
self.unload_models_supported = False
|
||||
self._client = client
|
||||
self._process: Optional[AudioCppServerProcess] = None
|
||||
self._lock = threading.RLock()
|
||||
self._closed = False
|
||||
self._model_ready_reported = False
|
||||
self.ui_session_id = uuid.uuid4().hex
|
||||
self._proxy = AudioCppRuntimeProxy(self) if self.owned else None
|
||||
|
||||
@property
|
||||
def process(self) -> Optional[AudioCppServerProcess]:
|
||||
return self._process
|
||||
|
||||
@property
|
||||
def running(self) -> bool:
|
||||
if not self.owned:
|
||||
return not self._closed
|
||||
return self._process is not None and self._process.running
|
||||
|
||||
@property
|
||||
def proxy(self) -> Optional[AudioCppRuntimeProxy]:
|
||||
return self._proxy
|
||||
|
||||
def _probe_opt_in_features(self) -> None:
|
||||
if not bool(self.config.get("probe_unload_models", False)) or self._client is None:
|
||||
return
|
||||
try:
|
||||
self.unload_models_supported = self._client.supports_feature("unload_models")
|
||||
except Exception as exc:
|
||||
_warn("audio.cpp unload_models feature probe failed", exc)
|
||||
self.unload_models_supported = False
|
||||
|
||||
def _start_owned_runtime(self) -> None:
|
||||
if not self.owned:
|
||||
return
|
||||
with _RUNTIME_START_LOCK:
|
||||
if self._process is not None and self._process.running:
|
||||
return
|
||||
_stop_conflicting_audio_cpp_sessions(self)
|
||||
if str(self.config.get("backend", "cuda")).lower() in {"cuda", "hip", "auto"}:
|
||||
_clear_conflicting_suite_tts_models()
|
||||
process_config = dict(self.config)
|
||||
process_config["model_id"] = self.model_id
|
||||
family = self.family or str(self.config.get("family", "unknown"))
|
||||
backend = str(self.config.get("backend", "auto"))
|
||||
print(
|
||||
f"🚀 audio.cpp: Starting {backend} server for {family} "
|
||||
f"('{self.model_id}')..."
|
||||
)
|
||||
started = time.monotonic()
|
||||
process = AudioCppServerProcess(process_config)
|
||||
process.start()
|
||||
self._process = process
|
||||
self._client = process.client
|
||||
self.endpoint = process.base_url
|
||||
self._probe_opt_in_features()
|
||||
if self._proxy is not None:
|
||||
self._proxy.register()
|
||||
print(
|
||||
f"✅ audio.cpp: Server ready at {self.endpoint} "
|
||||
f"({time.monotonic() - started:.2f}s); model will load on first request"
|
||||
)
|
||||
|
||||
def _ensure_client(self) -> AudioCppClient:
|
||||
if self._closed:
|
||||
raise RuntimeError("audio.cpp session is closed")
|
||||
if self.owned:
|
||||
if self._process is None or not self._process.running or self._client is None:
|
||||
if self._proxy is not None:
|
||||
self._proxy.unregister()
|
||||
if self._process is not None:
|
||||
self._process.close()
|
||||
self._process = None
|
||||
self._client = None
|
||||
self._start_owned_runtime()
|
||||
if self._client is None:
|
||||
raise RuntimeError("audio.cpp session has no HTTP client")
|
||||
return self._client
|
||||
|
||||
def _restart_after_transport_failure(self) -> AudioCppClient:
|
||||
if not self.owned:
|
||||
raise RuntimeError("Cannot restart an external audio.cpp server")
|
||||
self._stop_owned_runtime()
|
||||
self._start_owned_runtime()
|
||||
if self._client is None:
|
||||
raise RuntimeError("audio.cpp owned runtime restart did not create a client")
|
||||
return self._client
|
||||
|
||||
def restart_owned_runtime(self) -> None:
|
||||
"""Recreate an owned server when an upstream session cannot be reused safely."""
|
||||
if not self.owned:
|
||||
raise RuntimeError("Cannot restart an external audio.cpp server")
|
||||
with self._lock:
|
||||
self._stop_owned_runtime()
|
||||
self._start_owned_runtime()
|
||||
|
||||
def run(self, request: Mapping[str, Any]) -> AudioCppTaskResult:
|
||||
if not isinstance(request, Mapping):
|
||||
raise TypeError("audio.cpp task request must be a mapping")
|
||||
normalized_request = _normalize_request_paths(request)
|
||||
timeout = float(_first(self.config, "request_timeout", "request_timeout_seconds", default=600.0))
|
||||
with self._lock:
|
||||
client = self._ensure_client()
|
||||
first_request = not self._model_ready_reported
|
||||
started = time.monotonic()
|
||||
if first_request:
|
||||
action = "Loading model" if self.owned else "Sending first request to model"
|
||||
print(f"⏳ audio.cpp: {action} '{self.model_id}'...")
|
||||
try:
|
||||
result = client.run_task(self.model_id, normalized_request, timeout=timeout)
|
||||
except AudioCppConnectionError:
|
||||
if not self.owned:
|
||||
raise
|
||||
client = self._restart_after_transport_failure()
|
||||
result = client.run_task(self.model_id, normalized_request, timeout=timeout)
|
||||
except AudioCppTimeoutError:
|
||||
# A live server may still be executing after the client times out.
|
||||
# Only restart when the exact child has actually exited.
|
||||
if not self.owned or (self._process is not None and self._process.running):
|
||||
raise
|
||||
client = self._restart_after_transport_failure()
|
||||
result = client.run_task(self.model_id, normalized_request, timeout=timeout)
|
||||
if first_request:
|
||||
self._model_ready_reported = True
|
||||
print(
|
||||
f"✅ audio.cpp: Model '{self.model_id}' loaded; first {self.task} request completed "
|
||||
f"in {time.monotonic() - started:.2f}s"
|
||||
)
|
||||
return result
|
||||
|
||||
def voices(self) -> list[str]:
|
||||
timeout = float(_first(self.config, "connect_timeout", "connect_timeout_seconds", default=5.0))
|
||||
with self._lock:
|
||||
client = self._ensure_client()
|
||||
try:
|
||||
return client.voices(self.model_id, timeout=timeout)
|
||||
except AudioCppConnectionError:
|
||||
if not self.owned:
|
||||
raise
|
||||
return self._restart_after_transport_failure().voices(
|
||||
self.model_id, timeout=timeout
|
||||
)
|
||||
|
||||
def _stop_owned_runtime(self, *, unregister: bool = True) -> None:
|
||||
if not self.owned:
|
||||
return
|
||||
with self._lock:
|
||||
process = self._process
|
||||
self._process = None
|
||||
self._client = None
|
||||
self._model_ready_reported = False
|
||||
if unregister and self._proxy is not None:
|
||||
self._proxy.unregister()
|
||||
if process is not None:
|
||||
process.close()
|
||||
|
||||
def close(self) -> None:
|
||||
with self._lock:
|
||||
if self._closed:
|
||||
return
|
||||
self._closed = True
|
||||
if self.owned:
|
||||
self._stop_owned_runtime()
|
||||
else:
|
||||
# External servers are never unloaded, reconfigured, or terminated.
|
||||
self._client = None
|
||||
|
||||
|
||||
_SESSIONS: Dict[str, AudioCppSession] = {}
|
||||
_SESSIONS_LOCK = threading.RLock()
|
||||
_RUNTIME_START_LOCK = threading.RLock()
|
||||
|
||||
|
||||
def _stop_conflicting_audio_cpp_sessions(active: AudioCppSession) -> None:
|
||||
if str(active.config.get("backend", "cuda")).lower() not in {"cuda", "hip", "auto"}:
|
||||
return
|
||||
with _SESSIONS_LOCK:
|
||||
conflicts = [
|
||||
session
|
||||
for session in _SESSIONS.values()
|
||||
if session is not active
|
||||
and session.owned
|
||||
and session.running
|
||||
and str(session.config.get("backend", "cuda")).lower()
|
||||
in {"cuda", "hip", "auto"}
|
||||
]
|
||||
for session in conflicts:
|
||||
session._stop_owned_runtime()
|
||||
|
||||
|
||||
def _clear_conflicting_suite_tts_models() -> None:
|
||||
"""Clear known suite-managed TTS resources only when their modules are already live."""
|
||||
interface_module = sys.modules.get("utils.models.unified_model_interface")
|
||||
interface = getattr(interface_module, "unified_model_interface", None)
|
||||
if interface is not None:
|
||||
try:
|
||||
isolated = getattr(interface, "_isolated_model_cache", None)
|
||||
remover = getattr(interface, "_remove_isolated_model", None)
|
||||
if isinstance(isolated, dict) and callable(remover):
|
||||
for cache_key in list(isolated):
|
||||
if "_tts_" in cache_key:
|
||||
remover(cache_key)
|
||||
except Exception as exc:
|
||||
_warn("Could not clear a conflicting isolated TTS runtime", exc)
|
||||
|
||||
wrapper_module = sys.modules.get("utils.models.comfyui_model_wrapper")
|
||||
manager = getattr(wrapper_module, "tts_model_manager", None)
|
||||
cache = getattr(manager, "_model_cache", None)
|
||||
remover = getattr(manager, "remove_model", None)
|
||||
if isinstance(cache, dict) and callable(remover):
|
||||
try:
|
||||
for cache_key, wrapper in list(cache.items()):
|
||||
model_info = getattr(wrapper, "model_info", None)
|
||||
if getattr(model_info, "model_type", None) == "tts":
|
||||
remover(cache_key)
|
||||
except Exception as exc:
|
||||
_warn("Could not clear a conflicting embedded TTS model", exc)
|
||||
|
||||
|
||||
def _external_client(
|
||||
config: Mapping[str, Any],
|
||||
) -> tuple[AudioCppClient, str, str, Dict[str, Any]]:
|
||||
endpoint = _canonical_url(_first(config, "server_url", "endpoint", "base_url"))
|
||||
if not _is_loopback_url(endpoint) and not bool(config.get("allow_remote_server", False)):
|
||||
raise ValueError(
|
||||
"External audio.cpp servers must use loopback by default. "
|
||||
"Set allow_remote_server only when transport security and path access are understood."
|
||||
)
|
||||
client = AudioCppClient(
|
||||
endpoint,
|
||||
connect_timeout=float(
|
||||
_first(config, "connect_timeout", "connect_timeout_seconds", default=5.0)
|
||||
),
|
||||
request_timeout=float(
|
||||
_first(config, "request_timeout", "request_timeout_seconds", default=600.0)
|
||||
),
|
||||
)
|
||||
models = client.models()
|
||||
requested_id = _first(config, "model_id", "server_model_id")
|
||||
available_models = [dict(item) for item in models if item.get("id") is not None]
|
||||
available_ids = [str(item["id"]) for item in available_models]
|
||||
if requested_id is not None:
|
||||
model_id = str(requested_id)
|
||||
if model_id not in available_ids:
|
||||
raise ValueError(
|
||||
f"External audio.cpp server does not expose model '{model_id}'. "
|
||||
f"Available: {', '.join(available_ids) or '(none)'}"
|
||||
)
|
||||
elif len(available_ids) == 1:
|
||||
model_id = available_ids[0]
|
||||
elif not available_ids:
|
||||
raise ValueError("External audio.cpp server does not expose any configured models")
|
||||
else:
|
||||
raise ValueError(
|
||||
"External audio.cpp server exposes multiple models; select model_id explicitly"
|
||||
)
|
||||
selected_metadata = next(
|
||||
(item for item in available_models if str(item.get("id")) == model_id),
|
||||
{"id": model_id},
|
||||
)
|
||||
return client, endpoint, model_id, selected_metadata
|
||||
|
||||
|
||||
def get_audio_cpp_session(config: Mapping[str, Any]) -> AudioCppSession:
|
||||
"""Return a keyed persistent audio.cpp session for a loose engine config dict."""
|
||||
flattened = resolve_audio_cpp_config(_flatten_config(config))
|
||||
mode = _connection_mode(flattened)
|
||||
if mode == "external_server":
|
||||
endpoint_hint = _canonical_url(
|
||||
_first(flattened, "server_url", "endpoint", "base_url")
|
||||
)
|
||||
model_hint = _first(flattened, "model_id", "server_model_id")
|
||||
auto_key = None
|
||||
if isinstance(model_hint, str) and model_hint.strip():
|
||||
hinted_key = _session_key(flattened, mode, model_hint.strip(), endpoint_hint)
|
||||
with _SESSIONS_LOCK:
|
||||
existing = _SESSIONS.get(hinted_key)
|
||||
if existing is not None and not existing._closed:
|
||||
return existing
|
||||
else:
|
||||
# An omitted model id means "select the server's sole model". Cache
|
||||
# that resolution per endpoint so every text chunk does not repeat
|
||||
# /v1/models before reaching the already-persistent session.
|
||||
auto_key = _session_key(flattened, mode, "", endpoint_hint)
|
||||
with _SESSIONS_LOCK:
|
||||
existing = _SESSIONS.get(auto_key)
|
||||
if existing is not None and not existing._closed:
|
||||
return existing
|
||||
client, endpoint, model_id, model_metadata = _external_client(flattened)
|
||||
key = _session_key(flattened, mode, model_id, endpoint)
|
||||
with _SESSIONS_LOCK:
|
||||
existing = _SESSIONS.get(key)
|
||||
if existing is not None and not existing._closed:
|
||||
if auto_key is not None:
|
||||
_SESSIONS[auto_key] = existing
|
||||
return existing
|
||||
session = AudioCppSession(
|
||||
flattened,
|
||||
owned=False,
|
||||
model_id=model_id,
|
||||
client=client,
|
||||
endpoint=endpoint,
|
||||
model_metadata=model_metadata,
|
||||
)
|
||||
session._probe_opt_in_features()
|
||||
_SESSIONS[key] = session
|
||||
if auto_key is not None:
|
||||
_SESSIONS[auto_key] = session
|
||||
return session
|
||||
|
||||
model_id = _safe_model_id(flattened)
|
||||
key = _session_key(flattened, mode, model_id)
|
||||
with _SESSIONS_LOCK:
|
||||
existing = _SESSIONS.get(key)
|
||||
if existing is not None and not existing._closed:
|
||||
return existing
|
||||
session = AudioCppSession(flattened, owned=True, model_id=model_id)
|
||||
_SESSIONS[key] = session
|
||||
return session
|
||||
|
||||
|
||||
def close_all_audio_cpp_sessions() -> None:
|
||||
with _SESSIONS_LOCK:
|
||||
sessions = list({id(session): session for session in _SESSIONS.values()}.values())
|
||||
_SESSIONS.clear()
|
||||
for session in sessions:
|
||||
try:
|
||||
session.close()
|
||||
except Exception as exc:
|
||||
_warn("Failed to close audio.cpp session during shutdown", exc)
|
||||
|
||||
|
||||
def audio_cpp_session_statuses() -> list[Dict[str, Any]]:
|
||||
"""Return a path-free, side-effect-free snapshot for the frontend indicator."""
|
||||
with _SESSIONS_LOCK:
|
||||
sessions = list({id(session): session for session in _SESSIONS.values()}.values())
|
||||
statuses = []
|
||||
for session in sessions:
|
||||
if session._closed:
|
||||
continue
|
||||
if session.owned:
|
||||
if session.running:
|
||||
state = "model_ready" if session._model_ready_reported else "server_ready"
|
||||
else:
|
||||
state = "configured"
|
||||
else:
|
||||
state = "model_ready" if session._model_ready_reported else "server_ready"
|
||||
statuses.append(
|
||||
{
|
||||
"session_id": session.ui_session_id,
|
||||
"state": state,
|
||||
"owned": session.owned,
|
||||
"family": session.family,
|
||||
"model_id": session.model_id,
|
||||
"endpoint": session.endpoint if not session.owned else "",
|
||||
"pid": session.process.process.pid if session.owned and session.running else None,
|
||||
**_process_memory_status(
|
||||
session.process.process.pid if session.owned and session.running else None
|
||||
),
|
||||
}
|
||||
)
|
||||
return statuses
|
||||
|
||||
|
||||
def _process_memory_status(pid: Optional[int]) -> Dict[str, Optional[int]]:
|
||||
if not pid:
|
||||
return {"working_set_bytes": None, "private_bytes": None}
|
||||
try:
|
||||
import psutil
|
||||
|
||||
info = psutil.Process(pid).memory_info()
|
||||
return {
|
||||
"working_set_bytes": int(info.rss),
|
||||
"private_bytes": int(getattr(info, "private", info.vms)),
|
||||
}
|
||||
except Exception:
|
||||
return {"working_set_bytes": None, "private_bytes": None}
|
||||
|
||||
|
||||
def stop_owned_audio_cpp_session(session_id: str) -> bool:
|
||||
"""Stop, but retain, an exact Suite-owned session for lazy restart."""
|
||||
with _SESSIONS_LOCK:
|
||||
sessions = list({id(session): session for session in _SESSIONS.values()}.values())
|
||||
session = next((item for item in sessions if item.ui_session_id == session_id), None)
|
||||
if session is None:
|
||||
return False
|
||||
if not session.owned:
|
||||
raise PermissionError("External audio.cpp servers cannot be stopped by the Suite")
|
||||
session._stop_owned_runtime()
|
||||
return True
|
||||
|
||||
|
||||
atexit.register(close_all_audio_cpp_sessions)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"AudioCppRuntimeProxy",
|
||||
"AudioCppSession",
|
||||
"audio_cpp_session_statuses",
|
||||
"close_all_audio_cpp_sessions",
|
||||
"get_audio_cpp_session",
|
||||
"stop_owned_audio_cpp_session",
|
||||
]
|
||||
@@ -0,0 +1,175 @@
|
||||
"""Machine-local audio.cpp settings stored outside workflow JSON."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Mapping, Optional, Tuple
|
||||
|
||||
|
||||
SETTINGS_SCHEMA_VERSION = 1
|
||||
SETTINGS_FILENAME = "settings.json"
|
||||
|
||||
|
||||
class SettingsError(ValueError):
|
||||
"""Raised for invalid audio.cpp machine-local settings."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AudioCppSettings:
|
||||
schema_version: int = SETTINGS_SCHEMA_VERSION
|
||||
connection_mode: str = "managed"
|
||||
external_server_url: str = ""
|
||||
executable_path: str = ""
|
||||
model_roots: Tuple[str, ...] = ()
|
||||
managed_model_root: str = ""
|
||||
runtime_root: str = ""
|
||||
runtime_backend: str = "auto"
|
||||
host: str = "127.0.0.1"
|
||||
port: int = 0
|
||||
extras: Mapping[str, Any] = field(default_factory=dict, repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def from_mapping(cls, values: Mapping[str, Any]) -> "AudioCppSettings":
|
||||
if not isinstance(values, Mapping):
|
||||
raise SettingsError("audio.cpp settings must be a JSON object")
|
||||
known = {
|
||||
"schema_version",
|
||||
"connection_mode",
|
||||
"external_server_url",
|
||||
"executable_path",
|
||||
"model_roots",
|
||||
"managed_model_root",
|
||||
"runtime_root",
|
||||
"runtime_backend",
|
||||
"host",
|
||||
"port",
|
||||
}
|
||||
roots = values.get("model_roots", ())
|
||||
if roots is None:
|
||||
roots = ()
|
||||
if not isinstance(roots, (list, tuple)) or not all(isinstance(item, str) for item in roots):
|
||||
raise SettingsError("model_roots must be a list of paths")
|
||||
mode = str(values.get("connection_mode", "managed")).strip().lower()
|
||||
if mode not in {"managed", "external"}:
|
||||
raise SettingsError("connection_mode must be 'managed' or 'external'")
|
||||
backend = str(values.get("runtime_backend", "auto")).strip().lower()
|
||||
if backend not in {"auto", "cpu", "cuda"}:
|
||||
raise SettingsError("runtime_backend must be 'auto', 'cpu', or 'cuda'")
|
||||
try:
|
||||
port = int(values.get("port", 0))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise SettingsError("port must be an integer") from exc
|
||||
if not 0 <= port <= 65535:
|
||||
raise SettingsError("port must be between 0 and 65535")
|
||||
schema_version = int(values.get("schema_version", SETTINGS_SCHEMA_VERSION))
|
||||
if schema_version > SETTINGS_SCHEMA_VERSION:
|
||||
raise SettingsError(
|
||||
f"Unsupported audio.cpp settings schema {schema_version}; "
|
||||
f"maximum is {SETTINGS_SCHEMA_VERSION}"
|
||||
)
|
||||
return cls(
|
||||
schema_version=SETTINGS_SCHEMA_VERSION,
|
||||
connection_mode=mode,
|
||||
external_server_url=str(values.get("external_server_url", "")).strip(),
|
||||
executable_path=str(values.get("executable_path", "")).strip(),
|
||||
model_roots=tuple(item.strip() for item in roots if item.strip()),
|
||||
managed_model_root=str(values.get("managed_model_root", "")).strip(),
|
||||
runtime_root=str(values.get("runtime_root", "")).strip(),
|
||||
runtime_backend=backend,
|
||||
host=str(values.get("host", "127.0.0.1")).strip() or "127.0.0.1",
|
||||
port=port,
|
||||
extras={key: value for key, value in values.items() if key not in known},
|
||||
)
|
||||
|
||||
def to_mapping(self) -> Dict[str, Any]:
|
||||
values = dict(self.extras)
|
||||
serialized = asdict(self)
|
||||
serialized.pop("extras", None)
|
||||
serialized["model_roots"] = list(self.model_roots)
|
||||
values.update(serialized)
|
||||
return values
|
||||
|
||||
|
||||
def _import_folder_paths():
|
||||
try:
|
||||
import folder_paths # type: ignore
|
||||
|
||||
return folder_paths
|
||||
except (ImportError, RuntimeError):
|
||||
return None
|
||||
|
||||
|
||||
def _fallback_settings_directory() -> Path:
|
||||
if os.name == "nt" and os.environ.get("LOCALAPPDATA"):
|
||||
return Path(os.environ["LOCALAPPDATA"]) / "TTS Audio Suite" / "audio_cpp"
|
||||
if os.environ.get("XDG_CONFIG_HOME"):
|
||||
return Path(os.environ["XDG_CONFIG_HOME"]) / "tts_audio_suite" / "audio_cpp"
|
||||
return Path.home() / ".config" / "tts_audio_suite" / "audio_cpp"
|
||||
|
||||
|
||||
def get_settings_path(folder_paths_module=None) -> Path:
|
||||
"""Resolve settings below ComfyUI's internal user directory when available."""
|
||||
|
||||
module = folder_paths_module if folder_paths_module is not None else _import_folder_paths()
|
||||
if module is not None and hasattr(module, "get_system_user_directory"):
|
||||
try:
|
||||
base = Path(module.get_system_user_directory("tts_audio_suite"))
|
||||
return base / "audio_cpp" / SETTINGS_FILENAME
|
||||
except (OSError, TypeError, ValueError):
|
||||
pass
|
||||
return _fallback_settings_directory() / SETTINGS_FILENAME
|
||||
|
||||
|
||||
def load_settings(
|
||||
path: Optional[Path] = None,
|
||||
*,
|
||||
folder_paths_module=None,
|
||||
strict: bool = False,
|
||||
) -> AudioCppSettings:
|
||||
settings_path = Path(path) if path is not None else get_settings_path(folder_paths_module)
|
||||
if not settings_path.is_file():
|
||||
return AudioCppSettings()
|
||||
try:
|
||||
values = json.loads(settings_path.read_text(encoding="utf-8"))
|
||||
return AudioCppSettings.from_mapping(values)
|
||||
except (OSError, json.JSONDecodeError, SettingsError, TypeError, ValueError):
|
||||
if strict:
|
||||
raise
|
||||
# A damaged local preference file must not stop ComfyUI from loading.
|
||||
return AudioCppSettings()
|
||||
|
||||
|
||||
def save_settings(
|
||||
settings: AudioCppSettings,
|
||||
path: Optional[Path] = None,
|
||||
*,
|
||||
folder_paths_module=None,
|
||||
) -> Path:
|
||||
"""Atomically write settings in the same directory as the final file."""
|
||||
|
||||
if not isinstance(settings, AudioCppSettings):
|
||||
settings = AudioCppSettings.from_mapping(settings) # type: ignore[arg-type]
|
||||
settings_path = Path(path) if path is not None else get_settings_path(folder_paths_module)
|
||||
settings_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
descriptor, temporary_name = tempfile.mkstemp(
|
||||
prefix=f".{settings_path.name}.", suffix=".tmp", dir=settings_path.parent
|
||||
)
|
||||
temporary_path = Path(temporary_name)
|
||||
try:
|
||||
with os.fdopen(descriptor, "w", encoding="utf-8", newline="\n") as handle:
|
||||
json.dump(settings.to_mapping(), handle, indent=2, sort_keys=True, ensure_ascii=False)
|
||||
handle.write("\n")
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
os.replace(temporary_path, settings_path)
|
||||
except BaseException:
|
||||
try:
|
||||
temporary_path.unlink(missing_ok=True)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
return settings_path
|
||||
@@ -0,0 +1,73 @@
|
||||
"""Windows Job Object ownership for suite-launched native servers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
|
||||
class WindowsKillOnCloseJob:
|
||||
"""Kill assigned children when the owning Python process loses this handle."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.handle = None
|
||||
def assign(self, process_handle: int) -> None:
|
||||
if os.name != "nt":
|
||||
return
|
||||
import ctypes
|
||||
from ctypes import wintypes
|
||||
|
||||
class IO_COUNTERS(ctypes.Structure):
|
||||
_fields_ = [(name, ctypes.c_ulonglong) for name in (
|
||||
"ReadOperationCount", "WriteOperationCount", "OtherOperationCount",
|
||||
"ReadTransferCount", "WriteTransferCount", "OtherTransferCount",
|
||||
)]
|
||||
|
||||
class JOBOBJECT_BASIC_LIMIT_INFORMATION(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("PerProcessUserTimeLimit", ctypes.c_longlong),
|
||||
("PerJobUserTimeLimit", ctypes.c_longlong),
|
||||
("LimitFlags", wintypes.DWORD),
|
||||
("MinimumWorkingSetSize", ctypes.c_size_t),
|
||||
("MaximumWorkingSetSize", ctypes.c_size_t),
|
||||
("ActiveProcessLimit", wintypes.DWORD),
|
||||
("Affinity", ctypes.c_size_t),
|
||||
("PriorityClass", wintypes.DWORD),
|
||||
("SchedulingClass", wintypes.DWORD),
|
||||
]
|
||||
|
||||
class JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("BasicLimitInformation", JOBOBJECT_BASIC_LIMIT_INFORMATION),
|
||||
("IoInfo", IO_COUNTERS),
|
||||
("ProcessMemoryLimit", ctypes.c_size_t),
|
||||
("JobMemoryLimit", ctypes.c_size_t),
|
||||
("PeakProcessMemoryUsed", ctypes.c_size_t),
|
||||
("PeakJobMemoryUsed", ctypes.c_size_t),
|
||||
]
|
||||
|
||||
kernel32 = ctypes.WinDLL("kernel32", use_last_error=True)
|
||||
kernel32.CreateJobObjectW.restype = wintypes.HANDLE
|
||||
kernel32.SetInformationJobObject.argtypes = [
|
||||
wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD
|
||||
]
|
||||
kernel32.AssignProcessToJobObject.argtypes = [wintypes.HANDLE, wintypes.HANDLE]
|
||||
handle = kernel32.CreateJobObjectW(None, None)
|
||||
if not handle:
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
info = JOBOBJECT_EXTENDED_LIMIT_INFORMATION()
|
||||
info.BasicLimitInformation.LimitFlags = 0x00002000 # KILL_ON_JOB_CLOSE
|
||||
if not kernel32.SetInformationJobObject(handle, 9, ctypes.byref(info), ctypes.sizeof(info)):
|
||||
kernel32.CloseHandle(handle)
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
if not kernel32.AssignProcessToJobObject(handle, wintypes.HANDLE(process_handle)):
|
||||
kernel32.CloseHandle(handle)
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
self.handle = handle
|
||||
|
||||
def close(self) -> None:
|
||||
if self.handle is None or os.name != "nt":
|
||||
return
|
||||
import ctypes
|
||||
|
||||
ctypes.WinDLL("kernel32", use_last_error=True).CloseHandle(self.handle)
|
||||
self.handle = None
|
||||
@@ -118,11 +118,11 @@ PARAMETER_ENGINES = {
|
||||
'chatterbox', 'chatterbox_official_23lang', 'f5tts', 'higgs_audio',
|
||||
'higgs_audio_v3', 'vibevoice', 'index_tts', 'step_audio_editx', 'cosyvoice', 'qwen3_tts',
|
||||
'dots_tts', 'fish_audio_s2', 'omnivoice',
|
||||
'echo_tts', 'moss_tts', 'moss_soundeffect_v2', 'dramabox'
|
||||
'echo_tts', 'moss_tts', 'moss_soundeffect_v2', 'dramabox', 'audio_cpp'
|
||||
},
|
||||
'temperature': {
|
||||
'chatterbox', 'chatterbox_official_23lang', 'f5tts', 'higgs_audio',
|
||||
'higgs_audio_v3', 'vibevoice', 'index_tts', 'step_audio_editx', 'qwen3_tts', 'moss_tts', 'fish_audio_s2'
|
||||
'higgs_audio_v3', 'vibevoice', 'index_tts', 'step_audio_editx', 'qwen3_tts', 'moss_tts', 'fish_audio_s2', 'audio_cpp'
|
||||
},
|
||||
'cfg': {
|
||||
'f5tts', 'vibevoice', 'index_tts', 'chatterbox', 'chatterbox_official_23lang',
|
||||
@@ -150,10 +150,10 @@ PARAMETER_ENGINES = {
|
||||
'dramabox'
|
||||
},
|
||||
'num_steps': {
|
||||
'echo_tts', 'dots_tts', 'omnivoice'
|
||||
'echo_tts', 'dots_tts', 'omnivoice', 'audio_cpp'
|
||||
},
|
||||
'guidance_scale': {
|
||||
'dots_tts', 'omnivoice'
|
||||
'dots_tts', 'omnivoice', 'audio_cpp'
|
||||
},
|
||||
'duration': {
|
||||
'omnivoice'
|
||||
@@ -216,13 +216,13 @@ PARAMETER_ENGINES = {
|
||||
'chatterbox', 'chatterbox_official_23lang'
|
||||
},
|
||||
'speed': {
|
||||
'f5tts', 'cosyvoice', 'omnivoice'
|
||||
'f5tts', 'cosyvoice', 'omnivoice', 'audio_cpp'
|
||||
},
|
||||
'top_p': {
|
||||
'higgs_audio', 'higgs_audio_v3', 'vibevoice', 'index_tts', 'qwen3_tts', 'moss_tts', 'fish_audio_s2'
|
||||
'higgs_audio', 'higgs_audio_v3', 'vibevoice', 'index_tts', 'qwen3_tts', 'moss_tts', 'fish_audio_s2', 'audio_cpp'
|
||||
},
|
||||
'top_k': {
|
||||
'higgs_audio', 'higgs_audio_v3', 'index_tts', 'qwen3_tts', 'moss_tts'
|
||||
'higgs_audio', 'higgs_audio_v3', 'index_tts', 'qwen3_tts', 'moss_tts', 'audio_cpp'
|
||||
},
|
||||
'audio_temperature': {
|
||||
'moss_tts'
|
||||
@@ -234,7 +234,7 @@ PARAMETER_ENGINES = {
|
||||
'moss_tts'
|
||||
},
|
||||
'repetition_penalty': {
|
||||
'moss_tts', 'fish_audio_s2'
|
||||
'moss_tts', 'fish_audio_s2', 'audio_cpp'
|
||||
},
|
||||
'audio_repetition_penalty': {
|
||||
'moss_tts'
|
||||
@@ -243,7 +243,7 @@ PARAMETER_ENGINES = {
|
||||
'moss_tts'
|
||||
},
|
||||
'max_new_tokens': {
|
||||
'higgs_audio_v3', 'moss_tts', 'fish_audio_s2'
|
||||
'higgs_audio_v3', 'moss_tts', 'fish_audio_s2', 'audio_cpp'
|
||||
},
|
||||
'max_generate_length': {
|
||||
'dots_tts'
|
||||
@@ -252,7 +252,7 @@ PARAMETER_ENGINES = {
|
||||
'moss_tts'
|
||||
},
|
||||
'instruction': {
|
||||
'moss_tts'
|
||||
'moss_tts', 'audio_cpp'
|
||||
},
|
||||
'quality': {
|
||||
'moss_tts'
|
||||
@@ -364,7 +364,10 @@ PARAMETER_NODE_KEYS = {
|
||||
'ref_duration': 'ref_duration',
|
||||
'rescale_scale': 'rescale_scale',
|
||||
'prompt_template': 'prompt_template',
|
||||
'num_steps': 'num_steps',
|
||||
'num_steps': {
|
||||
'default': 'num_steps',
|
||||
'audio_cpp': 'num_inference_steps',
|
||||
},
|
||||
'guidance_scale': 'guidance_scale',
|
||||
'duration': 'duration',
|
||||
't_shift': 't_shift',
|
||||
@@ -386,7 +389,10 @@ PARAMETER_NODE_KEYS = {
|
||||
'speaker_kv_min_t': 'speaker_kv_min_t',
|
||||
'sequence_length': 'sequence_length',
|
||||
'exaggeration': 'exaggeration',
|
||||
'speed': 'speed',
|
||||
'speed': {
|
||||
'default': 'speed',
|
||||
'audio_cpp': 'speaking_rate',
|
||||
},
|
||||
'top_p': 'top_p',
|
||||
'top_k': 'top_k',
|
||||
'audio_temperature': 'audio_temperature',
|
||||
@@ -395,10 +401,16 @@ PARAMETER_NODE_KEYS = {
|
||||
'repetition_penalty': 'repetition_penalty',
|
||||
'audio_repetition_penalty': 'audio_repetition_penalty',
|
||||
'duration_tokens': 'duration_tokens',
|
||||
'max_new_tokens': 'max_new_tokens',
|
||||
'max_new_tokens': {
|
||||
'default': 'max_new_tokens',
|
||||
'audio_cpp': 'max_tokens',
|
||||
},
|
||||
'max_generate_length': 'max_generate_length',
|
||||
'n_vq_for_inference': 'n_vq_for_inference',
|
||||
'instruction': 'instruction',
|
||||
'instruction': {
|
||||
'default': 'instruction',
|
||||
'audio_cpp': 'instruct',
|
||||
},
|
||||
'quality': 'quality',
|
||||
'sound_event': 'sound_event',
|
||||
'ambient_sound': 'ambient_sound',
|
||||
|
||||
@@ -0,0 +1,460 @@
|
||||
import { app } from "../../scripts/app.js";
|
||||
import { api } from "../../scripts/api.js";
|
||||
|
||||
const TARGET = "AudioCppEngineNode";
|
||||
const ENDPOINT = "/api/tts-audio-suite/audio-cpp-capabilities";
|
||||
const STATUS_ENDPOINT = "/api/tts-audio-suite/audio-cpp-status";
|
||||
const PANEL_HEIGHT = 210;
|
||||
const PANEL_MIN_WIDTH = 360;
|
||||
const PANEL_BOTTOM_PADDING = 14;
|
||||
const PANEL_LAYOUT_HEIGHT = PANEL_HEIGHT + PANEL_BOTTOM_PADDING;
|
||||
const REQUEST_ADVANCED_WIDGETS = [
|
||||
"temperature", "top_p", "top_k", "repetition_penalty",
|
||||
"max_tokens", "max_steps", "num_inference_steps", "guidance_scale",
|
||||
"advanced_json",
|
||||
];
|
||||
const OWNED_ADVANCED_WIDGETS = [
|
||||
"auto_download_runtime", "auto_download_model", "show_server_console",
|
||||
];
|
||||
let manifestPromise;
|
||||
|
||||
function manifest() {
|
||||
manifestPromise ??= api.fetchApi(ENDPOINT).then((response) => {
|
||||
if (!response.ok) throw new Error(`Capability request failed (${response.status})`);
|
||||
return response.json();
|
||||
});
|
||||
return manifestPromise;
|
||||
}
|
||||
|
||||
function widget(node, name) {
|
||||
return (node.widgets || []).find((item) => item.name === name);
|
||||
}
|
||||
|
||||
function hideWidget(item) {
|
||||
if (!item || item.__ttsAudioCppHidden) return;
|
||||
item.__ttsAudioCppOriginalType = item.type;
|
||||
item.__ttsAudioCppOriginalComputeSize = item.computeSize;
|
||||
item.type = "hidden";
|
||||
item.hidden = true;
|
||||
item.computeSize = () => [0, -4];
|
||||
if (item.element) item.element.style.display = "none";
|
||||
item.__ttsAudioCppHidden = true;
|
||||
}
|
||||
|
||||
function showWidget(item) {
|
||||
if (!item || !item.__ttsAudioCppHidden) return;
|
||||
item.type = item.__ttsAudioCppOriginalType;
|
||||
item.computeSize = item.__ttsAudioCppOriginalComputeSize;
|
||||
item.hidden = false;
|
||||
if (item.element) item.element.style.display = "";
|
||||
item.__ttsAudioCppHidden = false;
|
||||
}
|
||||
|
||||
function setWidgetVisible(node, name, visible) {
|
||||
const item = widget(node, name);
|
||||
if (visible) showWidget(item);
|
||||
else hideWidget(item);
|
||||
}
|
||||
|
||||
function resizeNodeToContent(node) {
|
||||
if (node.__ttsAudioCppResizeFrame) cancelAnimationFrame(node.__ttsAudioCppResizeFrame);
|
||||
node.__ttsAudioCppResizeFrame = requestAnimationFrame(() => {
|
||||
node.__ttsAudioCppResizeFrame = 0;
|
||||
const computed = node.computeSize();
|
||||
const width = Math.max(Number(node.size?.[0]) || 0, PANEL_MIN_WIDTH, Number(computed?.[0]) || 0);
|
||||
const height = Number(computed?.[1]) || Number(node.size?.[1]) || PANEL_LAYOUT_HEIGHT;
|
||||
if (Math.abs(width - node.size[0]) > 0.5 || Math.abs(height - node.size[1]) > 0.5) {
|
||||
node.setSize([width, height]);
|
||||
}
|
||||
app.graph?.setDirtyCanvas(true, true);
|
||||
});
|
||||
}
|
||||
|
||||
function applyWidgetVisibility(node, capability) {
|
||||
if (!capability) return;
|
||||
const advanced = Boolean(node.__ttsAudioCppAdvancedOpen);
|
||||
const mode = String(widget(node, "connection_mode")?.value || "auto");
|
||||
const backend = String(widget(node, "backend")?.value || "auto");
|
||||
const external = mode === "external_server";
|
||||
const existingBinary = mode === "existing_binary";
|
||||
const suiteTasks = new Set(capability.suite_tasks || []);
|
||||
const upstreamTasks = new Set(capability.upstream_tasks || []);
|
||||
|
||||
setWidgetVisible(node, "package_id", !external);
|
||||
setWidgetVisible(node, "task", external || upstreamTasks.size > 1 || advanced);
|
||||
setWidgetVisible(node, "backend", !external);
|
||||
setWidgetVisible(node, "device", !external && backend !== "cpu" && advanced);
|
||||
setWidgetVisible(node, "threads", !external && (backend === "cpu" || advanced));
|
||||
setWidgetVisible(node, "language", suiteTasks.has("tts") || suiteTasks.has("asr"));
|
||||
setWidgetVisible(node, "server_url", external);
|
||||
setWidgetVisible(node, "binary_path", existingBinary || (!external && advanced));
|
||||
setWidgetVisible(node, "model_path", existingBinary || (!external && advanced));
|
||||
setWidgetVisible(node, "model_id", external);
|
||||
setWidgetVisible(node, "voice_id", Boolean(capability.built_in_voices));
|
||||
setWidgetVisible(node, "instruct", Boolean(capability.voice_design));
|
||||
for (const name of REQUEST_ADVANCED_WIDGETS) setWidgetVisible(node, name, advanced);
|
||||
for (const name of OWNED_ADVANCED_WIDGETS) setWidgetVisible(node, name, !external && advanced);
|
||||
|
||||
const toggle = widget(node, "audio_cpp_advanced_toggle");
|
||||
if (toggle) {
|
||||
toggle.label = advanced ? "▾ Hide advanced settings" : "▸ Show advanced settings";
|
||||
}
|
||||
resizeNodeToContent(node);
|
||||
}
|
||||
|
||||
function addAdvancedToggle(node) {
|
||||
const toggle = node.addWidget("button", "▸ Show advanced settings", null, () => {
|
||||
node.__ttsAudioCppAdvancedOpen = !node.__ttsAudioCppAdvancedOpen;
|
||||
applyWidgetVisibility(node, node.__ttsAudioCppCapability);
|
||||
});
|
||||
toggle.name = "audio_cpp_advanced_toggle";
|
||||
toggle.label = "▸ Show advanced settings";
|
||||
toggle.options ??= {};
|
||||
toggle.options.tooltip = "Show uncommon runtime paths, device tuning, sampling overrides, download controls, server debugging, and raw request JSON.";
|
||||
toggle.options.serialize = false;
|
||||
toggle.tooltip = toggle.options.tooltip;
|
||||
toggle.serialize = false;
|
||||
toggle.serializeValue = () => undefined;
|
||||
return toggle;
|
||||
}
|
||||
|
||||
function formatBytes(bytes) {
|
||||
const value = Number(bytes);
|
||||
if (!Number.isFinite(value) || value <= 0) return "unavailable";
|
||||
if (value >= 1073741824) return `${(value / 1073741824).toFixed(2)} GB`;
|
||||
return `${(value / 1048576).toFixed(1)} MB`;
|
||||
}
|
||||
|
||||
function syncPackageChoices(node, data, capability) {
|
||||
const packageWidget = widget(node, "package_id");
|
||||
if (!packageWidget || !capability) return null;
|
||||
const allowed = ["auto", ...(capability.packages || [])];
|
||||
packageWidget.options ??= {};
|
||||
packageWidget.options.values = allowed;
|
||||
if (!allowed.includes(String(packageWidget.value))) packageWidget.value = "auto";
|
||||
const resolvedId = packageWidget.value === "auto"
|
||||
? capability.recommended_package_id
|
||||
: String(packageWidget.value);
|
||||
return data.packages?.[resolvedId] || null;
|
||||
}
|
||||
|
||||
function syncTaskChoices(node, capability) {
|
||||
const taskWidget = widget(node, "task");
|
||||
if (!taskWidget || !capability) return;
|
||||
const allowed = ["auto", ...(capability.upstream_tasks || [])];
|
||||
taskWidget.options ??= {};
|
||||
taskWidget.options.values = [...new Set(allowed)];
|
||||
if (!taskWidget.options.values.includes(String(taskWidget.value))) taskWidget.value = "auto";
|
||||
}
|
||||
|
||||
function normalizedUrl(value) {
|
||||
return String(value || "").trim().replace(/\/$/, "");
|
||||
}
|
||||
|
||||
function matchingStatus(node, sessions) {
|
||||
const family = String(widget(node, "family")?.value || "");
|
||||
const modelId = String(widget(node, "model_id")?.value || "").trim();
|
||||
const mode = String(widget(node, "connection_mode")?.value || "auto");
|
||||
const endpoint = normalizedUrl(widget(node, "server_url")?.value);
|
||||
return (sessions || []).find((session) => {
|
||||
if (modelId && session.model_id !== modelId) return false;
|
||||
if (mode === "external_server" && endpoint) {
|
||||
return !session.owned && normalizedUrl(session.endpoint) === endpoint;
|
||||
}
|
||||
return session.family === family && (mode === "external_server" ? !session.owned : true);
|
||||
});
|
||||
}
|
||||
|
||||
function renderStatus(node, panel, sessions, failed = false) {
|
||||
const status = matchingStatus(node, sessions);
|
||||
const light = panel.querySelector(".tts-acpp-light");
|
||||
const label = panel.querySelector(".tts-acpp-status-text");
|
||||
const state = failed ? "error" : (status?.state || "unchecked");
|
||||
const labels = {
|
||||
unchecked: "Not checked",
|
||||
configured: "Server stopped",
|
||||
server_ready: "Server connected · model not confirmed",
|
||||
model_ready: "Server connected · model ready",
|
||||
error: "Status unavailable",
|
||||
};
|
||||
light.dataset.state = state;
|
||||
label.textContent = labels[state] || state;
|
||||
const memory = panel.querySelector(".tts-acpp-memory");
|
||||
const stop = panel.querySelector(".tts-acpp-stop");
|
||||
node.__ttsAudioCppSessionId = status?.session_id || "";
|
||||
if (status?.pid) {
|
||||
const privateGb = status.private_bytes ? ` · ${(status.private_bytes / 1073741824).toFixed(1)} GB private` : "";
|
||||
memory.textContent = `PID ${status.pid}${privateGb}`;
|
||||
memory.hidden = false;
|
||||
} else {
|
||||
memory.hidden = true;
|
||||
}
|
||||
stop.hidden = !(status?.owned && ["server_ready", "model_ready"].includes(status.state));
|
||||
}
|
||||
|
||||
async function refreshStatus(node, panel) {
|
||||
try {
|
||||
const response = await api.fetchApi(STATUS_ENDPOINT);
|
||||
if (!response.ok) throw new Error(`Status request failed (${response.status})`);
|
||||
const data = await response.json();
|
||||
renderStatus(node, panel, data.sessions);
|
||||
} catch (_error) {
|
||||
renderStatus(node, panel, [], true);
|
||||
}
|
||||
}
|
||||
|
||||
function inputIsSpeaker(input) {
|
||||
return /^speaker\d+$/.test(String(input?.name || ""));
|
||||
}
|
||||
|
||||
function removeUnusedSpeakerInputs(node, maximum) {
|
||||
for (let index = (node.inputs || []).length - 1; index >= 0; index -= 1) {
|
||||
const input = node.inputs[index];
|
||||
const number = Number(String(input?.name || "").replace("speaker", ""));
|
||||
if (inputIsSpeaker(input) && input.link == null && number > maximum) node.removeInput(index);
|
||||
}
|
||||
}
|
||||
|
||||
function syncSpeakers(node, capability) {
|
||||
const native = capability?.native_multi_speaker || {};
|
||||
const maximum = native.supported ? Number(native.max_speakers || 1) : 1;
|
||||
removeUnusedSpeakerInputs(node, maximum);
|
||||
const speakers = (node.inputs || []).filter(inputIsSpeaker);
|
||||
if (maximum <= 1) {
|
||||
for (let index = (node.inputs || []).length - 1; index >= 0; index -= 1) {
|
||||
if (inputIsSpeaker(node.inputs[index]) && node.inputs[index].link == null) node.removeInput(index);
|
||||
}
|
||||
return;
|
||||
}
|
||||
speakers.forEach((input, index) => {
|
||||
input.name = `speaker${index + 2}`;
|
||||
input.label = `Speaker ${index + 2}`;
|
||||
});
|
||||
const last = speakers[speakers.length - 1];
|
||||
if (speakers.length < maximum - 1 && (!last || last.link != null)) {
|
||||
node.addInput(`speaker${speakers.length + 2}`, "*");
|
||||
}
|
||||
}
|
||||
|
||||
function pill(text, tone = "normal", title = "") {
|
||||
const element = document.createElement("span");
|
||||
element.textContent = text;
|
||||
element.className = `tts-acpp-pill ${tone}`;
|
||||
if (title) element.title = title;
|
||||
return element;
|
||||
}
|
||||
|
||||
function appendAsrFeaturePills(container, capability) {
|
||||
const features = capability.asr_features || {};
|
||||
if (features.diarization === "native") {
|
||||
container.append(pill(
|
||||
"Diarization",
|
||||
"special",
|
||||
"Native speaker-attributed turns are available from this ASR family.",
|
||||
));
|
||||
} else {
|
||||
container.append(pill(
|
||||
"No diarization",
|
||||
"muted",
|
||||
"This ASR family does not return speaker identities.",
|
||||
));
|
||||
}
|
||||
|
||||
const timing = features.timing || "none";
|
||||
if (timing === "native_word") {
|
||||
container.append(pill(
|
||||
"Word timestamps",
|
||||
"info",
|
||||
"Native word or token timestamps are available without a separate aligner.",
|
||||
));
|
||||
} else if (timing === "native_segment") {
|
||||
container.append(pill(
|
||||
"Segment timestamps",
|
||||
"info",
|
||||
"Native timed transcript or speaker segments are available.",
|
||||
));
|
||||
} else if (timing === "optional_forced_aligner") {
|
||||
container.append(pill(
|
||||
"Optional forced aligner",
|
||||
"warn",
|
||||
"Word timestamps require the separate Qwen3 Forced Aligner model.",
|
||||
));
|
||||
} else {
|
||||
container.append(pill(
|
||||
"No timestamps",
|
||||
"muted",
|
||||
"This ASR family currently returns transcription text without timing alignment.",
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
function setWidgetHeight(widget, height) {
|
||||
try { widget.height = height; } catch (_error) { /* getter-only on some builds */ }
|
||||
try { widget.computedHeight = height; } catch (_error) { /* getter-only on some builds */ }
|
||||
}
|
||||
|
||||
function makePanel(node) {
|
||||
const panel = document.createElement("div");
|
||||
panel.className = "tts-acpp-panel";
|
||||
panel.innerHTML = `<style>
|
||||
.tts-acpp-panel{box-sizing:border-box;width:100%;max-width:100%;height:${PANEL_HEIGHT}px;overflow:hidden;margin:0;padding:9px 10px;border:1px solid var(--border-color,#454545);border-radius:7px;background:color-mix(in srgb,var(--comfy-menu-bg,#202020) 90%,#4d78a8 10%);color:var(--input-text,#ddd);font:12px/1.35 sans-serif}
|
||||
.tts-acpp-head{display:flex;flex-wrap:wrap;justify-content:space-between;align-items:center;gap:5px 8px;margin-bottom:5px}.tts-acpp-title{font-weight:650;font-size:13px}.tts-acpp-status{display:flex;align-items:center;gap:5px;color:var(--descrip-text,#aaa);font-size:11px;white-space:nowrap}
|
||||
.tts-acpp-light{width:8px;height:8px;border-radius:50%;background:#777;box-shadow:0 0 0 2px color-mix(in srgb,#777 25%,transparent)}.tts-acpp-light[data-state="configured"]{background:#d49a36}.tts-acpp-light[data-state="server_ready"]{background:#4d9bea}.tts-acpp-light[data-state="model_ready"]{background:#48bf78;box-shadow:0 0 5px #48bf78}.tts-acpp-light[data-state="error"]{background:#d85b5b}
|
||||
.tts-acpp-runtime{display:flex;flex-wrap:wrap;align-items:center;justify-content:space-between;gap:4px 8px;margin:-1px 0 6px;color:var(--descrip-text,#aaa);font-size:11px}.tts-acpp-stop{border:1px solid #6d4b4b;border-radius:5px;background:#3d2929;color:#efc5c5;padding:2px 6px;cursor:pointer}.tts-acpp-stop:hover{background:#543232}
|
||||
.tts-acpp-pills{display:flex;flex-wrap:wrap;gap:4px;margin-bottom:6px}
|
||||
.tts-acpp-pill{padding:2px 7px;border:1px solid transparent;border-radius:999px;background:#3a4652;color:#dcecff;font-size:11px;line-height:1.35;white-space:nowrap}.tts-acpp-pill.warn{border-color:#806635;background:#5b4929;color:#ffe0a3}.tts-acpp-pill.good{border-color:#38684e;background:#294d3b;color:#bdebd2}.tts-acpp-pill.info{border-color:#365f7d;background:#29485f;color:#c8e7ff}.tts-acpp-pill.special{border-color:#685485;background:#46385d;color:#e8d9ff}.tts-acpp-pill.muted{border-color:#4d565f;background:#343b42;color:#b9c1c9}
|
||||
.tts-acpp-detail{color:var(--descrip-text,#b8b8b8);margin-top:2px}.tts-acpp-summary{margin-top:6px;color:var(--input-text,#ddd)}
|
||||
</style><div class="tts-acpp-head"><div class="tts-acpp-title">audio.cpp capabilities</div><div class="tts-acpp-status"><span class="tts-acpp-light" data-state="unchecked"></span><span class="tts-acpp-status-text">Not checked</span></div></div><div class="tts-acpp-runtime"><span class="tts-acpp-memory" hidden></span><button class="tts-acpp-stop" type="button" hidden>Stop owned server</button></div><div class="tts-acpp-pills"></div><div class="tts-acpp-details"></div><div class="tts-acpp-summary"></div>`;
|
||||
panel.querySelector(".tts-acpp-stop").addEventListener("click", async () => {
|
||||
const sessionId = node.__ttsAudioCppSessionId;
|
||||
if (!sessionId) return;
|
||||
const response = await api.fetchApi("/api/tts-audio-suite/audio-cpp-stop", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ session_id: sessionId }),
|
||||
});
|
||||
if (!response.ok) {
|
||||
const data = await response.json().catch(() => ({}));
|
||||
throw new Error(data.error || `Stop failed (${response.status})`);
|
||||
}
|
||||
refreshStatus(node, panel);
|
||||
});
|
||||
const panelWidget = node.addDOMWidget("audio_cpp_capabilities", "div", panel, {
|
||||
serialize: false,
|
||||
hideOnZoom: false,
|
||||
getMinHeight: () => PANEL_LAYOUT_HEIGHT,
|
||||
getHeight: () => PANEL_LAYOUT_HEIGHT,
|
||||
});
|
||||
panelWidget.computeSize = (inputWidth) => {
|
||||
const width = Array.isArray(inputWidth) ? inputWidth[0] : inputWidth;
|
||||
return [Math.max(PANEL_MIN_WIDTH, Number(width) || PANEL_MIN_WIDTH), PANEL_LAYOUT_HEIGHT];
|
||||
};
|
||||
panelWidget.getHeight = () => PANEL_LAYOUT_HEIGHT;
|
||||
panelWidget.computeLayoutSize = () => ({
|
||||
minWidth: PANEL_MIN_WIDTH,
|
||||
minHeight: PANEL_LAYOUT_HEIGHT,
|
||||
});
|
||||
panelWidget.options ??= {};
|
||||
panelWidget.options.minNodeSize = [PANEL_MIN_WIDTH, PANEL_LAYOUT_HEIGHT];
|
||||
setWidgetHeight(panelWidget, PANEL_LAYOUT_HEIGHT);
|
||||
if (panelWidget.element) {
|
||||
panelWidget.element.style.boxSizing = "border-box";
|
||||
panelWidget.element.style.width = "100%";
|
||||
panelWidget.element.style.maxWidth = "100%";
|
||||
panelWidget.element.style.height = `${PANEL_HEIGHT}px`;
|
||||
panelWidget.element.style.minHeight = `${PANEL_HEIGHT}px`;
|
||||
panelWidget.element.style.overflow = "hidden";
|
||||
}
|
||||
return panel;
|
||||
}
|
||||
|
||||
function render(node, panel, data, capability) {
|
||||
if (!capability) return;
|
||||
node.__ttsAudioCppCapability = capability;
|
||||
const selectedPackage = syncPackageChoices(node, data, capability);
|
||||
syncTaskChoices(node, capability);
|
||||
panel.querySelector(".tts-acpp-title").textContent = capability.display_name;
|
||||
const pills = panel.querySelector(".tts-acpp-pills");
|
||||
pills.replaceChildren();
|
||||
const suiteTasks = new Set(capability.suite_tasks || []);
|
||||
if (suiteTasks.has("tts")) pills.append(pill("TTS", "good", "Text-to-speech is wired to the Suite's Unified Text and SRT nodes."));
|
||||
if (suiteTasks.has("asr")) pills.append(pill("ASR", "good", "Speech recognition is wired to the Suite's Unified ASR node."));
|
||||
if (suiteTasks.has("voice_conversion")) pills.append(pill("Voice conversion", "good", "Voice conversion is wired to the Suite's Unified Voice Changer node."));
|
||||
if (suiteTasks.has("asr")) appendAsrFeaturePills(pills, capability);
|
||||
if (suiteTasks.has("tts") && capability.reference_audio !== "none") pills.append(pill("Voice clone", "normal", "This family accepts reference audio for voice cloning or conditioning."));
|
||||
if (["required", "required_per_speaker"].includes(capability.reference_audio)) {
|
||||
pills.append(pill("Reference required", "warn", "Generation requires reference audio."));
|
||||
}
|
||||
if (capability.reference_transcript === "required") {
|
||||
pills.append(pill("Transcript required", "warn", "The transcript matching the reference audio is required."));
|
||||
}
|
||||
if (capability.built_in_voices) pills.append(pill("Built-in voices", "info", "This family includes model-provided voices."));
|
||||
if (capability.voice_design) pills.append(pill("Voice design", "normal", "This family can synthesize from a written voice description."));
|
||||
if (capability.inline_controls) pills.append(pill("Inline controls", "normal", "Suite-standard inline controls are translated for this family."));
|
||||
if (capability.native_multi_speaker?.supported) {
|
||||
const status = capability.native_multi_speaker.suite_status === "supported" ? "good" : "warn";
|
||||
pills.append(pill(`Up to ${capability.native_multi_speaker.max_speakers} speakers`, status));
|
||||
}
|
||||
const transcript = capability.reference_transcript;
|
||||
let taskDetails = "";
|
||||
if (suiteTasks.has("tts")) {
|
||||
taskDetails =
|
||||
`<div class="tts-acpp-detail">Reference audio: ${capability.reference_audio.replaceAll("_", " ")}</div>` +
|
||||
`<div class="tts-acpp-detail">Reference transcript: ${transcript}</div>`;
|
||||
} else if (suiteTasks.has("asr")) {
|
||||
const languages = capability.languages || [];
|
||||
const languageSummary = languages.length > 8 ? `${languages.length} declared languages` : languages.join(", ");
|
||||
taskDetails = `<div class="tts-acpp-detail">Languages: ${languageSummary || "model-defined"}</div>`;
|
||||
} else if (suiteTasks.has("voice_conversion")) {
|
||||
taskDetails = `<div class="tts-acpp-detail">Inputs: source audio + target reference audio</div>`;
|
||||
}
|
||||
panel.querySelector(".tts-acpp-details").innerHTML =
|
||||
`<div class="tts-acpp-detail">Package: ${selectedPackage?.display_name || "external server / unresolved"}</div>` +
|
||||
`<div class="tts-acpp-detail">Estimated download: ${formatBytes(selectedPackage?.estimated_download_bytes)}</div>` +
|
||||
taskDetails +
|
||||
`<div class="tts-acpp-detail">Suite: ${(capability.suite_tasks || []).join(" · ")}</div>`;
|
||||
panel.querySelector(".tts-acpp-summary").textContent = capability.summary || capability.description;
|
||||
syncSpeakers(node, capability);
|
||||
applyWidgetVisibility(node, capability);
|
||||
}
|
||||
|
||||
function hookWidgetCallback(node, name, callback) {
|
||||
const item = widget(node, name);
|
||||
if (!item || item.__ttsAudioCppCallbackHooked) return;
|
||||
item.__ttsAudioCppCallbackHooked = true;
|
||||
const original = item.callback;
|
||||
item.callback = (...args) => {
|
||||
const result = original?.apply(item, args);
|
||||
callback();
|
||||
return result;
|
||||
};
|
||||
}
|
||||
|
||||
app.registerExtension({
|
||||
name: "TTS_Audio_Suite.AudioCppCapabilities",
|
||||
async beforeRegisterNodeDef(nodeType, nodeData) {
|
||||
if ((nodeData?.name || nodeData?.comfyClass) !== TARGET) return;
|
||||
const original = nodeType.prototype.onNodeCreated;
|
||||
nodeType.prototype.onNodeCreated = function() {
|
||||
const result = original?.apply(this, arguments);
|
||||
addAdvancedToggle(this);
|
||||
const panel = makePanel(this);
|
||||
const family = widget(this, "family");
|
||||
const update = () => manifest().then((data) => render(this, panel, data, data.families?.[family?.value])).catch((error) => {
|
||||
panel.querySelector(".tts-acpp-summary").textContent = error.message;
|
||||
});
|
||||
hookWidgetCallback(this, "family", update);
|
||||
hookWidgetCallback(this, "package_id", update);
|
||||
hookWidgetCallback(this, "task", update);
|
||||
hookWidgetCallback(this, "connection_mode", () => applyWidgetVisibility(this, this.__ttsAudioCppCapability));
|
||||
hookWidgetCallback(this, "backend", () => applyWidgetVisibility(this, this.__ttsAudioCppCapability));
|
||||
const refresh = () => refreshStatus(this, panel);
|
||||
this.__ttsAudioCppStatusRefresh = refresh;
|
||||
setTimeout(() => { update(); refresh(); }, 0);
|
||||
this.__ttsAudioCppStatusTimer = setInterval(refresh, 5000);
|
||||
return result;
|
||||
};
|
||||
const removed = nodeType.prototype.onRemoved;
|
||||
nodeType.prototype.onRemoved = function() {
|
||||
clearInterval(this.__ttsAudioCppStatusTimer);
|
||||
if (this.__ttsAudioCppResizeFrame) cancelAnimationFrame(this.__ttsAudioCppResizeFrame);
|
||||
return removed?.apply(this, arguments);
|
||||
};
|
||||
const connectionChanged = nodeType.prototype.onConnectionsChange;
|
||||
nodeType.prototype.onConnectionsChange = function(type, index) {
|
||||
const result = connectionChanged?.apply(this, arguments);
|
||||
if (inputIsSpeaker(this.inputs?.[index])) {
|
||||
const family = widget(this, "family")?.value;
|
||||
manifest().then((data) => syncSpeakers(this, data.families?.[family]));
|
||||
}
|
||||
return result;
|
||||
};
|
||||
},
|
||||
setup() {
|
||||
for (const eventName of ["executing", "executed", "execution_error"]) {
|
||||
api.addEventListener(eventName, () => {
|
||||
for (const node of app.graph?._nodes || []) node.__ttsAudioCppStatusRefresh?.();
|
||||
});
|
||||
}
|
||||
},
|
||||
});
|
||||
Reference in New Issue
Block a user