412 lines
8.9 KiB
JSON
412 lines
8.9 KiB
JSON
{
|
|
"id": "4a7cdb14-f535-4923-861b-59ba2c7ee043",
|
|
"revision": 0,
|
|
"last_node_id": 42,
|
|
"last_link_id": 83,
|
|
"nodes": [
|
|
{
|
|
"id": 35,
|
|
"type": "LoadAudio",
|
|
"pos": [
|
|
-1328.9635009765625,
|
|
2782.253173828125
|
|
],
|
|
"size": [
|
|
270,
|
|
136
|
|
],
|
|
"flags": {},
|
|
"order": 0,
|
|
"mode": 0,
|
|
"inputs": [],
|
|
"outputs": [
|
|
{
|
|
"name": "AUDIO",
|
|
"type": "AUDIO",
|
|
"links": [
|
|
72
|
|
]
|
|
}
|
|
],
|
|
"properties": {
|
|
"cnr_id": "comfy-core",
|
|
"ver": "0.3.39",
|
|
"Node name for S&R": "LoadAudio",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": [
|
|
"male_stewie.mp3",
|
|
null,
|
|
null
|
|
]
|
|
},
|
|
{
|
|
"id": 34,
|
|
"type": "LoadAudio",
|
|
"pos": [
|
|
-1333.963134765625,
|
|
3007.423095703125
|
|
],
|
|
"size": [
|
|
270,
|
|
136
|
|
],
|
|
"flags": {},
|
|
"order": 1,
|
|
"mode": 0,
|
|
"inputs": [],
|
|
"outputs": [
|
|
{
|
|
"name": "AUDIO",
|
|
"type": "AUDIO",
|
|
"links": [
|
|
73
|
|
]
|
|
}
|
|
],
|
|
"properties": {
|
|
"cnr_id": "comfy-core",
|
|
"ver": "0.3.39",
|
|
"Node name for S&R": "LoadAudio",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": [
|
|
"male_harvey_keitel.mp3",
|
|
null,
|
|
null
|
|
]
|
|
},
|
|
{
|
|
"id": 37,
|
|
"type": "ChatterBoxVoiceVC",
|
|
"pos": [
|
|
-939.0897216796875,
|
|
2909.46435546875
|
|
],
|
|
"size": [
|
|
305.8872985839844,
|
|
78
|
|
],
|
|
"flags": {},
|
|
"order": 4,
|
|
"mode": 0,
|
|
"inputs": [
|
|
{
|
|
"name": "source_audio",
|
|
"type": "AUDIO",
|
|
"link": 72
|
|
},
|
|
{
|
|
"name": "target_audio",
|
|
"type": "AUDIO",
|
|
"link": 73
|
|
}
|
|
],
|
|
"outputs": [
|
|
{
|
|
"name": "converted_audio",
|
|
"type": "AUDIO",
|
|
"links": [
|
|
74
|
|
]
|
|
}
|
|
],
|
|
"properties": {
|
|
"Node name for S&R": "ChatterBoxVoiceVC",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": [
|
|
"auto"
|
|
]
|
|
},
|
|
{
|
|
"id": 42,
|
|
"type": "ChatterBoxVoiceTTS",
|
|
"pos": [
|
|
-1100.9342041015625,
|
|
2206.41259765625
|
|
],
|
|
"size": [
|
|
400,
|
|
372
|
|
],
|
|
"flags": {},
|
|
"order": 5,
|
|
"mode": 0,
|
|
"inputs": [
|
|
{
|
|
"name": "reference_audio",
|
|
"shape": 7,
|
|
"type": "AUDIO",
|
|
"link": 82
|
|
},
|
|
{
|
|
"name": "text",
|
|
"type": "STRING",
|
|
"widget": {
|
|
"name": "text"
|
|
},
|
|
"link": 81
|
|
}
|
|
],
|
|
"outputs": [
|
|
{
|
|
"name": "audio",
|
|
"type": "AUDIO",
|
|
"links": [
|
|
83
|
|
]
|
|
},
|
|
{
|
|
"name": "generation_info",
|
|
"type": "STRING",
|
|
"links": null
|
|
}
|
|
],
|
|
"properties": {
|
|
"Node name for S&R": "ChatterBoxVoiceTTS",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": [
|
|
"Hello! This is the enhanced ChatterboxTTS with improved chunking. It can handle very long texts by intelligently splitting them into smaller segments.",
|
|
"auto",
|
|
0.5,
|
|
0.8,
|
|
0.5,
|
|
2354936645,
|
|
"randomize",
|
|
"",
|
|
true,
|
|
400,
|
|
"auto",
|
|
100
|
|
]
|
|
},
|
|
{
|
|
"id": 15,
|
|
"type": "PreviewAudio",
|
|
"pos": [
|
|
-517.1994018554688,
|
|
2329.54736328125
|
|
],
|
|
"size": [
|
|
270,
|
|
88
|
|
],
|
|
"flags": {},
|
|
"order": 7,
|
|
"mode": 0,
|
|
"inputs": [
|
|
{
|
|
"name": "audio",
|
|
"type": "AUDIO",
|
|
"link": 83
|
|
}
|
|
],
|
|
"outputs": [],
|
|
"properties": {
|
|
"cnr_id": "comfy-core",
|
|
"ver": "0.3.39",
|
|
"Node name for S&R": "PreviewAudio",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": []
|
|
},
|
|
{
|
|
"id": 36,
|
|
"type": "PreviewAudio",
|
|
"pos": [
|
|
-542.3068237304688,
|
|
2905.98583984375
|
|
],
|
|
"size": [
|
|
270,
|
|
88
|
|
],
|
|
"flags": {},
|
|
"order": 6,
|
|
"mode": 0,
|
|
"inputs": [
|
|
{
|
|
"name": "audio",
|
|
"type": "AUDIO",
|
|
"link": 74
|
|
}
|
|
],
|
|
"outputs": [],
|
|
"properties": {
|
|
"cnr_id": "comfy-core",
|
|
"ver": "0.3.39",
|
|
"Node name for S&R": "PreviewAudio",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": []
|
|
},
|
|
{
|
|
"id": 26,
|
|
"type": "LoadAudio",
|
|
"pos": [
|
|
-1413.8255615234375,
|
|
2431.993408203125
|
|
],
|
|
"size": [
|
|
270,
|
|
136
|
|
],
|
|
"flags": {},
|
|
"order": 2,
|
|
"mode": 0,
|
|
"inputs": [],
|
|
"outputs": [
|
|
{
|
|
"name": "AUDIO",
|
|
"type": "AUDIO",
|
|
"links": [
|
|
82
|
|
]
|
|
}
|
|
],
|
|
"properties": {
|
|
"cnr_id": "comfy-core",
|
|
"ver": "0.3.39",
|
|
"Node name for S&R": "LoadAudio",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": [
|
|
"female_shadowheart.flac",
|
|
null,
|
|
null
|
|
]
|
|
},
|
|
{
|
|
"id": 39,
|
|
"type": "String Literal",
|
|
"pos": [
|
|
-2198.8310546875,
|
|
2162.29931640625
|
|
],
|
|
"size": [
|
|
671.1473999023438,
|
|
1065.6341552734375
|
|
],
|
|
"flags": {},
|
|
"order": 3,
|
|
"mode": 0,
|
|
"inputs": [],
|
|
"outputs": [
|
|
{
|
|
"name": "STRING",
|
|
"type": "STRING",
|
|
"links": [
|
|
81
|
|
]
|
|
}
|
|
],
|
|
"properties": {
|
|
"cnr_id": "image-saver",
|
|
"ver": "65e6903eff274a50f8b5cd768f0f96baf37baea1",
|
|
"Node name for S&R": "String Literal",
|
|
"widget_ue_connectable": {}
|
|
},
|
|
"widgets_values": [
|
|
"We're excited to introduce Chatterbox, Resemble AI's first production-grade open source TTS model. Licensed under MIT, Chatterbox has been benchmarked against leading closed-source systems like ElevenLabs, and is consistently preferred in side-by-side evaluations.\n\nWhether you're working on memes, videos, games, or AI agents, Chatterbox brings your content to life. It's also the first open source TTS model to support emotion exaggeration control, a powerful feature that makes your voices stand out. Try it now on our Hugging Face Gradio app.\n\nIf you like the model but need to scale or tune it for higher accuracy, check out our competitively priced TTS service (link). It delivers reliable performance with ultra-low latency of sub 200ms—ideal for production use in agents, applications, or interactive media.\n\nKey Details\nSoTA zeroshot TTS\n0.5B Llama backbone\nUnique exaggeration/intensity control\nUltra-stable with alignment-informed inference\nTrained on 0.5M hours of cleaned data\nWatermarked outputs\nEasy voice conversion script\nOutperforms ElevenLabs\nTips\nGeneral Use (TTS and Voice Agents):\n\nThe default settings (exaggeration=0.5, cfg_weight=0.5) work well for most prompts.\nIf the reference speaker has a fast speaking style, lowering cfg_weight to around 0.3 can improve pacing.\nExpressive or Dramatic Speech:\n\nTry lower cfg_weight values (e.g. ~0.3) and increase exaggeration to around 0.7 or higher.\nHigher exaggeration tends to speed up speech; reducing cfg_weight helps compensate with slower, more deliberate pacing."
|
|
]
|
|
}
|
|
],
|
|
"links": [
|
|
[
|
|
72,
|
|
35,
|
|
0,
|
|
37,
|
|
0,
|
|
"AUDIO"
|
|
],
|
|
[
|
|
73,
|
|
34,
|
|
0,
|
|
37,
|
|
1,
|
|
"AUDIO"
|
|
],
|
|
[
|
|
74,
|
|
37,
|
|
0,
|
|
36,
|
|
0,
|
|
"AUDIO"
|
|
],
|
|
[
|
|
81,
|
|
39,
|
|
0,
|
|
42,
|
|
1,
|
|
"STRING"
|
|
],
|
|
[
|
|
82,
|
|
26,
|
|
0,
|
|
42,
|
|
0,
|
|
"AUDIO"
|
|
],
|
|
[
|
|
83,
|
|
42,
|
|
0,
|
|
15,
|
|
0,
|
|
"AUDIO"
|
|
]
|
|
],
|
|
"groups": [
|
|
{
|
|
"id": 2,
|
|
"title": "ChatterBox Text-to-Speech",
|
|
"bounding": [
|
|
-1513.438720703125,
|
|
2120.572021484375,
|
|
1376.1483154296875,
|
|
501.39434814453125
|
|
],
|
|
"color": "#3f789e",
|
|
"font_size": 24,
|
|
"flags": {}
|
|
},
|
|
{
|
|
"id": 3,
|
|
"title": "ChatterBox Voice Conversion",
|
|
"bounding": [
|
|
-1510.47412109375,
|
|
2652.15234375,
|
|
1371.289794921875,
|
|
575.38330078125
|
|
],
|
|
"color": "#3f789e",
|
|
"font_size": 24,
|
|
"flags": {}
|
|
}
|
|
],
|
|
"config": {},
|
|
"extra": {
|
|
"ue_links": [],
|
|
"ds": {
|
|
"scale": 0.7247295000000005,
|
|
"offset": [
|
|
2653.7854970832586,
|
|
-1965.6967349619167
|
|
]
|
|
},
|
|
"links_added_by_ue": [],
|
|
"frontendVersion": "1.21.6",
|
|
"VHS_latentpreview": false,
|
|
"VHS_latentpreviewrate": 0,
|
|
"VHS_MetadataImage": true,
|
|
"VHS_KeepIntermediate": true
|
|
},
|
|
"version": 0.4
|
|
} |