add gemini multi tts
This commit is contained in:
+10
-4
@@ -4,28 +4,34 @@ from typing_extensions import override
|
||||
from .gemini.gemini_image_node import *
|
||||
from .gemini.gemini_image_preset_node import *
|
||||
from .gemini.gemini_tts_node import *
|
||||
from .gemini.gemini_tts_multi_node import *
|
||||
from .gemini.gemini_stt_node import *
|
||||
from .ollama.ollama_vlm_node import *
|
||||
from .ollama.ollama_llm_node import *
|
||||
from .options.ollama_llm_advanced_options_node import *
|
||||
from .modelscope.modelscope_image_node import *
|
||||
from .options.config_options_node import *
|
||||
from .options.gemini_speaker_options_node import *
|
||||
from .options.gemini_batch_speakers_options_node import *
|
||||
from .options.proxy_options_node import *
|
||||
|
||||
class APIExtension(ComfyExtension):
|
||||
@override
|
||||
async def get_node_list(self) -> list[type[io.ComfyNode]]:
|
||||
return [
|
||||
GeminiImage,
|
||||
GeminiImagePreset,
|
||||
GeminiImage,
|
||||
GeminiTTS,
|
||||
GeminiTTSMulti,
|
||||
GeminiSTT,
|
||||
OllamaVLM,
|
||||
OllamaLLM,
|
||||
OllamaLLMAdvanceOptions,
|
||||
OllamaVLM,
|
||||
ModelScopeImage,
|
||||
ConfigOptions,
|
||||
ProxyOptions,
|
||||
GeminiImagePreset,
|
||||
GeminiSpeakerOptions,
|
||||
GeminiBatchSpeakersOptions,
|
||||
OllamaLLMAdvanceOptions,
|
||||
]
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
from comfy_api.latest import io
|
||||
|
||||
from .gemini_tts_node import GeminiTTS
|
||||
|
||||
|
||||
class GeminiTTSMulti(io.ComfyNode):
|
||||
"""
|
||||
这个节点使用谷歌 Gemini TTS API 生成人物对话语音
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def define_schema(cls) -> io.Schema:
|
||||
model_options = GeminiTTS._load_models_from_config()
|
||||
default_model = model_options[0]
|
||||
|
||||
return io.Schema(
|
||||
node_id="YCYY_Gemini_TTS_Multi_API",
|
||||
display_name="Gemini TTS Multi API",
|
||||
category="YCYY/API/audio",
|
||||
inputs=[
|
||||
io.String.Input(
|
||||
id="text",
|
||||
default="## THE SCENE\n设置场景的背景信息,包括地点、氛围和环境细节,以确定基调和氛围。\n" \
|
||||
"## DIRECTOR'S NOTES\n导演备注,仅定义对性能至关重要的内容,并注意不要过度指定。最常见的指令是风格、语速和口音,但模型不限于这些指令,也不要求使用这些指令。您可以随意添加自定义说明\n" \
|
||||
"## TRANSCRIPT\n转写内容和音频标记,转写内容是模型将要朗读的确切字词。音频标记是指方括号中的字词,用于指示说话方式、音调变化或插话。多说话人名称需要与配置对应,示例如下:\n" \
|
||||
"Speaker 1: I know right, I couldn't believe it. [whispers] She should have totally left at that point.\n" \
|
||||
"Speaker 2: [cough] Well, [sighs] I guess it doesn't matter now.",
|
||||
multiline=True,
|
||||
tooltip="Conversation text. Include speaker labels that match the configured speaker names."
|
||||
),
|
||||
io.AnyType.Input(
|
||||
id="config_options",
|
||||
optional=True,
|
||||
tooltip="Optional configuration override from YCYY Gemini TTS Config Options"
|
||||
),
|
||||
io.AnyType.Input(
|
||||
id="proxy_options",
|
||||
optional=True,
|
||||
tooltip="Optional proxy configuration override from YCYY Proxy Config Options"
|
||||
),
|
||||
io.AnyType.Input(
|
||||
id="speaker_options",
|
||||
tooltip="Speaker option array from Gemini Speaker Options or Gemini Batch Speakers Options"
|
||||
),
|
||||
io.Combo.Input(
|
||||
id="model",
|
||||
options=model_options,
|
||||
default=default_model
|
||||
),
|
||||
io.Int.Input(
|
||||
id="seed",
|
||||
min=0,
|
||||
max=0xFFFFFFFFFFFFFFFF,
|
||||
default=0,
|
||||
control_after_generate=True
|
||||
)
|
||||
],
|
||||
outputs=[
|
||||
io.Audio.Output(),
|
||||
io.String.Output()
|
||||
],
|
||||
description="This node uses the Google Gemini TTS API to generate multi-speaker speech from text."
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def execute(cls, text, speaker_options, model, seed, config_options=None, proxy_options=None) -> io.NodeOutput:
|
||||
base_url, api_key, timeout = GeminiTTS._load_config_credentials(config_options)
|
||||
proxies = GeminiTTS._get_proxy_config(proxy_options)
|
||||
|
||||
if not text:
|
||||
raise ValueError("text cannot be empty")
|
||||
|
||||
normalized_speaker_options = cls._normalize_speaker_options(speaker_options)
|
||||
if not normalized_speaker_options:
|
||||
raise ValueError("speaker_options cannot be empty")
|
||||
|
||||
api_url = base_url + "/" + model + ":generateContent"
|
||||
return cls._generate_speech(api_url, api_key, text, model, normalized_speaker_options, timeout, proxies)
|
||||
|
||||
@classmethod
|
||||
def _normalize_speaker_options(cls, speaker_options):
|
||||
if not isinstance(speaker_options, list):
|
||||
raise ValueError("speaker_options must be a list")
|
||||
|
||||
normalized = []
|
||||
seen_speakers = set()
|
||||
for item in speaker_options:
|
||||
if not isinstance(item, dict):
|
||||
raise ValueError("Each speaker option must be an object")
|
||||
|
||||
speaker = str(item.get("speaker", "")).strip()
|
||||
voice_name = str(item.get("voiceName", "")).strip()
|
||||
|
||||
if not speaker:
|
||||
raise ValueError("speaker cannot be empty")
|
||||
if not voice_name:
|
||||
raise ValueError("voiceName cannot be empty")
|
||||
if speaker in seen_speakers:
|
||||
raise ValueError(f"Duplicate speaker is not allowed: {speaker}")
|
||||
|
||||
seen_speakers.add(speaker)
|
||||
normalized.append({
|
||||
"speaker": speaker,
|
||||
"voiceName": voice_name,
|
||||
})
|
||||
|
||||
return normalized
|
||||
|
||||
@classmethod
|
||||
def _generate_speech(cls, api_url, api_key, text, model, speaker_options, timeout, proxies=None) -> io.NodeOutput:
|
||||
headers = {
|
||||
"x-goog-api-key": api_key,
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
|
||||
payload = {
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": text
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"generationConfig": {
|
||||
"responseModalities": ["AUDIO"],
|
||||
"speechConfig": {
|
||||
"multiSpeakerVoiceConfig": {
|
||||
"speakerVoiceConfigs": [
|
||||
{
|
||||
"speaker": item["speaker"],
|
||||
"voiceConfig": {
|
||||
"prebuiltVoiceConfig": {
|
||||
"voiceName": item["voiceName"]
|
||||
}
|
||||
}
|
||||
}
|
||||
for item in speaker_options
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
try:
|
||||
import requests
|
||||
resp = requests.post(api_url, headers=headers, json=payload, timeout=timeout, proxies=proxies)
|
||||
return GeminiTTS._parse_response(resp)
|
||||
except Exception as e:
|
||||
return io.NodeOutput(None, f'{{"success":false,"message":"The API request failed. Please check if the interface address and key are correct. Error: {str(e)}"}}')
|
||||
+35
-32
@@ -7,6 +7,40 @@ import numpy as np
|
||||
from comfy_api.latest import io
|
||||
|
||||
|
||||
VOICE_OPTIONS = [
|
||||
"Zephyr",
|
||||
"Puck",
|
||||
"Charon",
|
||||
"Kore",
|
||||
"Fenrir",
|
||||
"Leda",
|
||||
"Orus",
|
||||
"Aoede",
|
||||
"Callirrhoe",
|
||||
"Autonoe",
|
||||
"Enceladus",
|
||||
"Iapetus",
|
||||
"Umbriel",
|
||||
"Algieba",
|
||||
"Despina",
|
||||
"Erinome",
|
||||
"Algenib",
|
||||
"Rasalgethi",
|
||||
"Laomedeia",
|
||||
"Achernar",
|
||||
"Alnilam",
|
||||
"Schedar",
|
||||
"Gacrux",
|
||||
"Pulcherrima",
|
||||
"Achird",
|
||||
"Zubenelgenubi",
|
||||
"Vindemiatrix",
|
||||
"Sadachbia",
|
||||
"Sadaltager",
|
||||
"Sulafat",
|
||||
]
|
||||
|
||||
|
||||
class GeminiTTS(io.ComfyNode):
|
||||
"""
|
||||
这个节点使用谷歌Gemini TTS API 生成语音
|
||||
@@ -173,38 +207,7 @@ class GeminiTTS(io.ComfyNode):
|
||||
),
|
||||
io.Combo.Input(
|
||||
id="voiceName",
|
||||
options=[
|
||||
"Zephyr",
|
||||
"Puck",
|
||||
"Charon",
|
||||
"Kore",
|
||||
"Fenrir",
|
||||
"Leda",
|
||||
"Orus",
|
||||
"Aoede",
|
||||
"Callirrhoe",
|
||||
"Autonoe",
|
||||
"Enceladus",
|
||||
"Iapetus",
|
||||
"Umbriel",
|
||||
"Algieba",
|
||||
"Despina",
|
||||
"Erinome",
|
||||
"Algenib",
|
||||
"Rasalgethi",
|
||||
"Laomedeia",
|
||||
"Achernar",
|
||||
"Alnilam",
|
||||
"Schedar",
|
||||
"Gacrux",
|
||||
"Pulcherrima",
|
||||
"Achird",
|
||||
"Zubenelgenubi",
|
||||
"Vindemiatrix",
|
||||
"Sadachbia",
|
||||
"Sadaltager",
|
||||
"Sulafat"
|
||||
],
|
||||
options=VOICE_OPTIONS,
|
||||
default="Zephyr",
|
||||
tooltip="The voice to use for speech synthesis"
|
||||
),
|
||||
|
||||
+101
-24
@@ -46,29 +46,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Image_Preset": {
|
||||
"display_name": "Gemini Image Preset",
|
||||
"description": "This node provides presets for the Gemini Image API.",
|
||||
"inputs": {
|
||||
"preset": {
|
||||
"name": "preset",
|
||||
"tooltip": "Gemini image preset name"
|
||||
},
|
||||
"description": {
|
||||
"name": "description",
|
||||
"tooltip": "Gemini image preset description"
|
||||
},
|
||||
"prompt": {
|
||||
"name": "prompt",
|
||||
"tooltip": "Gemini image preset prompt"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "String"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Ollama_VLM_API": {
|
||||
"display_name": "Ollama VLM API",
|
||||
"description": "This node uses the Ollama VLM model for image reasoning and analysis.",
|
||||
@@ -225,6 +202,69 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Image_Preset": {
|
||||
"display_name": "Gemini Image Preset",
|
||||
"description": "This node provides presets for the Gemini Image API.",
|
||||
"inputs": {
|
||||
"preset": {
|
||||
"name": "preset",
|
||||
"tooltip": "Gemini image preset name"
|
||||
},
|
||||
"description": {
|
||||
"name": "description",
|
||||
"tooltip": "Gemini image preset description"
|
||||
},
|
||||
"prompt": {
|
||||
"name": "prompt",
|
||||
"tooltip": "Gemini image preset prompt"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "String"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Speaker_Options": {
|
||||
"display_name": "Gemini Speaker Options",
|
||||
"description": "This node builds a single speaker option for Gemini multi-speaker TTS.",
|
||||
"inputs": {
|
||||
"speaker": {
|
||||
"name": "speaker",
|
||||
"tooltip": "Speaker name used in the conversation text"
|
||||
},
|
||||
"voiceName": {
|
||||
"name": "voiceName",
|
||||
"tooltip": "The voice assigned to this speaker"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "speaker_options",
|
||||
"tooltip": "Single speaker option item for Gemini multi-speaker TTS"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Batch_Speakers_Options": {
|
||||
"display_name": "Gemini Batch Speakers Options",
|
||||
"description": "This node merges two Gemini multi-speaker option arrays.",
|
||||
"inputs": {
|
||||
"speaker_options1": {
|
||||
"name": "speaker_options1",
|
||||
"tooltip": "The first speaker options array"
|
||||
},
|
||||
"speaker_options2": {
|
||||
"name": "speaker_options2",
|
||||
"tooltip": "The second speaker options array"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "speaker_options",
|
||||
"tooltip": "Merged speaker options array"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_TTS_API": {
|
||||
"display_name": "Gemini TTS API",
|
||||
"description": "This node uses the Google Gemini TTS API to generate speech from text.",
|
||||
@@ -258,6 +298,43 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_TTS_Multi_API": {
|
||||
"display_name": "Gemini TTS Multi API",
|
||||
"description": "This node uses the Google Gemini TTS API to generate multi-speaker speech from text.",
|
||||
"inputs": {
|
||||
"text": {
|
||||
"name": "text",
|
||||
"tooltip": "Conversation text. Include speaker labels that match the configured speaker names."
|
||||
},
|
||||
"speaker_options": {
|
||||
"name": "speaker_options",
|
||||
"tooltip": "Speaker option array from Gemini Speaker Options or Gemini Batch Speakers Options"
|
||||
},
|
||||
"config_options": {
|
||||
"name": "config_options",
|
||||
"tooltip": "Optional configuration override from YCYY API Config Options"
|
||||
},
|
||||
"proxy_options": {
|
||||
"name": "proxy_options",
|
||||
"tooltip": "Optional proxy configuration override from YCYY API Proxy Options"
|
||||
},
|
||||
"model": {
|
||||
"name": "model"
|
||||
},
|
||||
"seed": {
|
||||
"name": "seed",
|
||||
"tooltip": "Seed to use for generation"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "Audio"
|
||||
},
|
||||
"1": {
|
||||
"name": "String"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_STT_API": {
|
||||
"display_name": "Gemini STT API",
|
||||
"description": "This node uses the Google Gemini STT API to transcribe speech to text.",
|
||||
@@ -336,4 +413,4 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+101
-24
@@ -47,29 +47,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Image_Preset":{
|
||||
"display_name": "Gemini 图像预设",
|
||||
"description": "该节点为 Gemini 图像 API 提供预设值",
|
||||
"inputs": {
|
||||
"preset":{
|
||||
"name": "预设名称",
|
||||
"tooltip": "预设名称"
|
||||
},
|
||||
"description":{
|
||||
"name": "预设描述",
|
||||
"tooltip": "预设描述"
|
||||
},
|
||||
"prompt":{
|
||||
"name": "预设提示词",
|
||||
"tooltip": "预设提示词"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "字符串"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Ollama_VLM_API":{
|
||||
"display_name": "Ollama 视觉 API",
|
||||
"description": "这个节点使用Ollama VLM 模型进行图片推理分析",
|
||||
@@ -226,6 +203,69 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Image_Preset":{
|
||||
"display_name": "Gemini 图像预设",
|
||||
"description": "该节点为 Gemini 图像 API 提供预设值",
|
||||
"inputs": {
|
||||
"preset":{
|
||||
"name": "预设名称",
|
||||
"tooltip": "预设名称"
|
||||
},
|
||||
"description":{
|
||||
"name": "预设描述",
|
||||
"tooltip": "预设描述"
|
||||
},
|
||||
"prompt":{
|
||||
"name": "预设提示词",
|
||||
"tooltip": "预设提示词"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "字符串"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Speaker_Options": {
|
||||
"display_name": "Gemini 说话人选项",
|
||||
"description": "该节点用于构造 Gemini 多说话人 TTS 的单个说话人配置",
|
||||
"inputs": {
|
||||
"speaker": {
|
||||
"name": "说话人名称",
|
||||
"tooltip": "对话文本中使用的说话人名称"
|
||||
},
|
||||
"voiceName": {
|
||||
"name": "语音名称",
|
||||
"tooltip": "分配给该说话人的语音"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "说话人选项",
|
||||
"tooltip": "Gemini 多说话人 TTS 的单个说话人选项"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_Batch_Speakers_Options": {
|
||||
"display_name": "Gemini 批量说话人选项",
|
||||
"description": "该节点用于合并两个 Gemini 多说话人选项数组",
|
||||
"inputs": {
|
||||
"speaker_options1": {
|
||||
"name": "说话人选项1",
|
||||
"tooltip": "第一个说话人选项数组"
|
||||
},
|
||||
"speaker_options2": {
|
||||
"name": "说话人选项2",
|
||||
"tooltip": "第二个说话人选项数组"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "说话人选项",
|
||||
"tooltip": "合并后的说话人选项数组"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_TTS_API": {
|
||||
"display_name": "Gemini 文本转语音 API",
|
||||
"description": "该节点使用 Google Gemini TTS API 将文本转换为语音",
|
||||
@@ -259,6 +299,43 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_TTS_Multi_API": {
|
||||
"display_name": "Gemini 多人文本转语音 API",
|
||||
"description": "该节点使用 Google Gemini TTS API 将文本转换为多说话人语音",
|
||||
"inputs": {
|
||||
"text": {
|
||||
"name": "text",
|
||||
"tooltip": "对话文本,需包含与配置说话人名称一致的说话人标签"
|
||||
},
|
||||
"speaker_options": {
|
||||
"name": "说话人选项",
|
||||
"tooltip": "来自 Gemini 说话人选项或 Gemini 批量说话人选项节点的说话人数组"
|
||||
},
|
||||
"config_options": {
|
||||
"name": "配置选项",
|
||||
"tooltip": "可选配置覆盖选项,来自 YCYY API 配置选项"
|
||||
},
|
||||
"proxy_options": {
|
||||
"name": "代理选项",
|
||||
"tooltip": "可选代理覆盖选项,来自 YCYY API 代理选项"
|
||||
},
|
||||
"model": {
|
||||
"name": "model"
|
||||
},
|
||||
"seed": {
|
||||
"name": "种子",
|
||||
"tooltip": "用于生成的随机种子"
|
||||
}
|
||||
},
|
||||
"outputs": {
|
||||
"0": {
|
||||
"name": "Audio"
|
||||
},
|
||||
"1": {
|
||||
"name": "String"
|
||||
}
|
||||
}
|
||||
},
|
||||
"YCYY_Gemini_STT_API": {
|
||||
"display_name": "Gemini 语音转文本 API",
|
||||
"description": "该节点使用 Google Gemini STT API 进行语音识别",
|
||||
@@ -337,4 +414,4 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
from comfy_api.latest import io
|
||||
|
||||
|
||||
class GeminiBatchSpeakersOptions(io.ComfyNode):
|
||||
"""
|
||||
这个节点用于合并两个 Gemini 说话人配置数组
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def define_schema(cls) -> io.Schema:
|
||||
return io.Schema(
|
||||
node_id="YCYY_Gemini_Batch_Speakers_Options",
|
||||
display_name="Gemini Batch Speakers Options",
|
||||
category="YCYY/API/utils",
|
||||
inputs=[
|
||||
io.AnyType.Input(
|
||||
id="speaker_options1",
|
||||
tooltip="The first speaker options array"
|
||||
),
|
||||
io.AnyType.Input(
|
||||
id="speaker_options2",
|
||||
tooltip="The second speaker options array"
|
||||
)
|
||||
],
|
||||
outputs=[
|
||||
io.AnyType.Output(
|
||||
id="speaker_options",
|
||||
display_name="speaker_options",
|
||||
tooltip="Merged speaker options array"
|
||||
)
|
||||
],
|
||||
description="This node merges two Gemini multi-speaker option arrays."
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def execute(cls, speaker_options1, speaker_options2) -> io.NodeOutput:
|
||||
merged_options = cls._normalize_options(speaker_options1) + cls._normalize_options(speaker_options2)
|
||||
if not merged_options:
|
||||
raise ValueError("speaker_options cannot be empty")
|
||||
|
||||
return io.NodeOutput(merged_options)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_options(value):
|
||||
if value is None:
|
||||
return []
|
||||
if not isinstance(value, list):
|
||||
raise ValueError("speaker_options input must be a list")
|
||||
|
||||
normalized = []
|
||||
for item in value:
|
||||
if not isinstance(item, dict):
|
||||
raise ValueError("Each speaker option must be an object")
|
||||
|
||||
speaker = str(item.get("speaker", "")).strip()
|
||||
voice_name = str(item.get("voiceName", "")).strip()
|
||||
|
||||
if not speaker:
|
||||
raise ValueError("speaker cannot be empty")
|
||||
if not voice_name:
|
||||
raise ValueError("voiceName cannot be empty")
|
||||
|
||||
normalized.append({
|
||||
"speaker": speaker,
|
||||
"voiceName": voice_name,
|
||||
})
|
||||
|
||||
return normalized
|
||||
@@ -0,0 +1,51 @@
|
||||
from comfy_api.latest import io
|
||||
|
||||
from ..gemini.gemini_tts_node import VOICE_OPTIONS
|
||||
|
||||
|
||||
class GeminiSpeakerOptions(io.ComfyNode):
|
||||
"""
|
||||
这个节点用于构造 Gemini 多说话人 TTS 的单个说话人配置
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def define_schema(cls) -> io.Schema:
|
||||
return io.Schema(
|
||||
node_id="YCYY_Gemini_Speaker_Options",
|
||||
display_name="Gemini Speaker Options",
|
||||
category="YCYY/API/utils",
|
||||
inputs=[
|
||||
io.String.Input(
|
||||
id="speaker",
|
||||
default="Speaker 1",
|
||||
tooltip="Speaker name used in the conversation text"
|
||||
),
|
||||
io.Combo.Input(
|
||||
id="voiceName",
|
||||
options=VOICE_OPTIONS,
|
||||
default="Zephyr",
|
||||
tooltip="The voice assigned to this speaker"
|
||||
)
|
||||
],
|
||||
outputs=[
|
||||
io.AnyType.Output(
|
||||
id="speaker_options",
|
||||
display_name="speaker_options",
|
||||
tooltip="Single speaker option item for Gemini multi-speaker TTS"
|
||||
)
|
||||
],
|
||||
description="This node builds a single speaker option for Gemini multi-speaker TTS."
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def execute(cls, speaker, voiceName) -> io.NodeOutput:
|
||||
speaker_name = speaker.strip() if speaker else ""
|
||||
if not speaker_name:
|
||||
raise ValueError("speaker cannot be empty")
|
||||
|
||||
return io.NodeOutput([
|
||||
{
|
||||
"speaker": speaker_name,
|
||||
"voiceName": voiceName,
|
||||
}
|
||||
])
|
||||
Reference in New Issue
Block a user