Compare commits

...
21 Commits
Author SHA1 Message Date
shadowcz007 96929b6d7c Update README.md 2024-10-12 20:40:06 +08:00
shadowcz007 07712d80a5 add SimulateDevDesignDiscussions 多智能体播客节点 2024-10-12 20:39:39 +08:00
shadow 10c9eff16f Merge pull request #348 from shadowcz007/whisper-sensevoice
Whisper sensevoice
2024-10-12 10:46:17 +08:00
shadowcz007 edd7af986d update 2024-10-12 10:44:39 +08:00
shadowcz007 1dc31927e3 Update extension-node-map.json 2024-10-12 10:43:15 +08:00
shadowcz007 36ef7d25ef Update ui_mixlab.js 2024-10-05 12:31:22 +08:00
shadowcz007 b766b8b65d Update SenseVoice.py 2024-10-03 11:13:14 +08:00
shadowcz007 6579ff20b4 json_string2 2024-10-03 11:12:22 +08:00
shadowcz007 2fbee59c3e fixbug 2024-10-03 11:12:10 +08:00
shadowcz007 d3aaa19148 Update ChatGPT.py 2024-10-02 17:07:10 +08:00
shadowcz007 e32a3675fc Update Whisper.py 2024-10-02 17:05:43 +08:00
shadowcz007 b72e7dda08 Update Audio.py 2024-10-02 17:05:35 +08:00
shadowcz007 0f77f28a95 Update Whisper.py 2024-10-02 16:41:46 +08:00
shadowcz007 289f83675b Update SenseVoice.py 2024-10-02 16:41:44 +08:00
shadowcz007 36633b4c72 update 2024-10-02 14:09:08 +08:00
shadowcz007 4f45457811 Update extension-node-map.json 2024-10-02 11:05:33 +08:00
shadowcz007 c39890cd64 MiniCPM_VQA_Simple add extract_keywords 2024-10-02 11:05:20 +08:00
shadowcz007 90f1e49263 Merge branch 'main' of https://github.com/shadowcz007/comfyui-mixlab-nodes 2024-10-02 09:34:42 +08:00
shadowcz007 9a1cf205db Update requirements.txt 2024-10-02 09:34:18 +08:00
shadow 45eacb6a50 Merge pull request #342 from shadowcz007/SenseVoice
Update requirements.txt
2024-10-02 09:23:38 +08:00
shadowcz007 6cb2b57463 Update requirements.txt 2024-10-02 09:23:15 +08:00
13 changed files with 2246 additions and 290 deletions
+2
View File
@@ -10,6 +10,8 @@ For business cooperation, please contact email 389570357@qq.com
##### `最新`:
- 新增 SimulateDevDesignDiscussions,需要安装[swarm](https://github.com/openai/swarm)和[Comfyui-ChatTTS](https://github.com/shadowcz007/Comfyui-ChatTTS),工作流下载[./workflow/swarm制作的播客节点workflow.json]
- 新增 SenseVoice
- [新增JS-SDK,方便直接在前端项目中使用comfyui](https://github.com/shadowcz007/comfyui-js-sdk)
+21 -6
View File
@@ -32,7 +32,7 @@ _URL_=None
# except:
# print("##nodes.ChatGPT ImportError")
from .nodes.ChatGPT import openai_client
# from .nodes.ChatGPT import openai_client
from .nodes.RembgNode import get_rembg_models,U2NET_HOME,run_briarmbg,run_rembg
@@ -1006,7 +1006,7 @@ from .nodes.ImageNode import DepthViewer_,ImageBatchToList_,ImageListToBatch_,Co
# from .nodes.Vae import VAELoader,VAEDecode
from .nodes.ScreenShareNode import ScreenShareNode,FloatingVideo
from .nodes.Audio import AudioPlayNode,SpeechRecognition,SpeechSynthesis
from .nodes.Audio import AudioPlayNode,SpeechRecognition,SpeechSynthesis,AnalyzeAudioNone
from .nodes.Utils import CreateJsonNode,KeyInput,IncrementingListNode,ListSplit,CreateLoraNames,CreateSampler_names,CreateCkptNames,CreateSeedNode,TESTNODE_,TESTNODE_TOKEN,AppInfo,IntNumber,FloatSlider,TextInput,ColorInput,FontInput,TextToNumber,DynamicDelayProcessor,LimitNumber,SwitchByIndex,MultiplicationNode
from .nodes.Mask import PreviewMask_,MaskListReplace,MaskListMerge,OutlineMask,FeatheredMask
@@ -1103,6 +1103,7 @@ NODE_CLASS_MAPPINGS = {
"SpeechRecognition":SpeechRecognition,
"SpeechSynthesis":SpeechSynthesis,
"AudioPlay":AudioPlayNode,
"AnalyzeAudio":AnalyzeAudioNone,
# Text
"TextToNumber":TextToNumber,
@@ -1220,6 +1221,7 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"SpeechSynthesis":"SpeechSynthesis ♾️Mixlab",
"SpeechRecognition":"SpeechRecognition ♾️Mixlab",
"AudioPlay":"Preview Audio ♾️Mixlab",
"AnalyzeAudio":"Analyze Audio ♾️Mixlab",
# Utils
"DynamicDelayProcessor":"DynamicDelayByText ♾️Mixlab",
@@ -1259,7 +1261,7 @@ logging.info('\033[91m ### Mixlab Nodes: \033[93mLoaded')
# print('\033[91m ### Mixlab Nodes: \033[93mLoaded')
try:
from .nodes.ChatGPT import SiliconflowTextToImageNode,JsonRepair,ChatGPTNode,ShowTextForGPT,CharacterInText,TextSplitByDelimiter,SiliconflowFreeNode
from .nodes.ChatGPT import SimulateDevDesignDiscussions,SiliconflowTextToImageNode,JsonRepair,ChatGPTNode,ShowTextForGPT,CharacterInText,TextSplitByDelimiter,SiliconflowFreeNode
logging.info('ChatGPT.available True')
NODE_CLASS_MAPPINGS_V = {
@@ -1269,7 +1271,9 @@ try:
"ShowTextForGPT":ShowTextForGPT,
"CharacterInText":CharacterInText,
"TextSplitByDelimiter":TextSplitByDelimiter,
"JsonRepair":JsonRepair
"JsonRepair":JsonRepair,
"SimulateDevDesignDiscussions":SimulateDevDesignDiscussions
}
# 一个包含节点友好/可读的标题的字典
@@ -1280,7 +1284,9 @@ try:
"ShowTextForGPT":"Show Text ♾️MixlabApp",
"CharacterInText":"Character In Text",
"TextSplitByDelimiter":"Text Split By Delimiter",
"JsonRepair":"Json Repair"
"JsonRepair":"Json Repair",
"SimulateDevDesignDiscussions":"SimulateDevDesignDiscussions ♾️Mixlab Podcast"
}
@@ -1433,11 +1439,20 @@ try:
from .nodes.SenseVoice import SenseVoiceNode
logging.info('SenseVoice.available')
NODE_CLASS_MAPPINGS['SenseVoiceNode']=SenseVoiceNode
NODE_DISPLAY_NAME_MAPPINGS["SenseVoiceNode"]= "Sense Voice"
NODE_DISPLAY_NAME_MAPPINGS["SenseVoiceNode"]= "Sense Voice ♾️Mixlab"
except Exception as e:
logging.info('SenseVoice.available False' )
try:
from .nodes.Whisper import LoadWhisperModel,WhisperTranscribe
logging.info('Whisper.available')
NODE_CLASS_MAPPINGS['LoadWhisperModel_']=LoadWhisperModel
NODE_CLASS_MAPPINGS['WhisperTranscribe_']=WhisperTranscribe
NODE_DISPLAY_NAME_MAPPINGS["LoadWhisperModel_"]= "Load Whisper Model ♾️Mixlab"
NODE_DISPLAY_NAME_MAPPINGS["WhisperTranscribe_"]= "Whisper Transcribe ♾️Mixlab"
except Exception as e:
logging.info('Whisper.available False' )
logging.info('\033[93m -------------- \033[0m')
+1191 -257
View File
File diff suppressed because it is too large Load Diff
+95
View File
@@ -3,6 +3,101 @@ import os
import folder_paths
import torchaudio
class AnyType(str):
"""A special class that is always equal in not equal comparisons. Credit to pythongosssss"""
def __ne__(self, __value: object) -> bool:
return False
any_type = AnyType("*")
def analyze_audio_data(audio_data):
total_duration = 0
total_gap_duration = 0
emotion_counts = {}
audio_types = set()
languages = set()
for i, entry in enumerate(audio_data):
# Calculate the duration of each audio segment
start_time = entry['start_time']
end_time = entry['end_time']
duration = end_time - start_time
total_duration += duration
# Count the emotions
if "emotion" in entry:
emotion = entry['emotion']
if emotion in emotion_counts:
emotion_counts[emotion] += 1
else:
emotion_counts[emotion] = 1
# Collect the audio types
if "audio_type" in entry:
audio_types.add(entry['audio_type'])
if "language" in entry:
languages.add(entry['language'])
# Calculate gap duration if not the last entry
if i < len(audio_data) - 1:
next_start_time = audio_data[i + 1]['start_time']
gap_duration = next_start_time - end_time
if gap_duration > 0:
total_gap_duration += gap_duration
# Get the most frequent emotion
if len(emotion_counts.keys())>0:
most_frequent_emotion = max(emotion_counts, key=emotion_counts.get)
else:
most_frequent_emotion=None
# Convert audio_types set to list for better readability
audio_types = list(audio_types)
languages=list(languages)
# Print the results
print(f"Total Effective Duration: {total_duration:.2f} seconds")
print(f"Total Gap Duration: {total_gap_duration:.2f} seconds")
print(f"Emotion Changes: {emotion_counts}")
print(f"Most Frequent Emotion: {most_frequent_emotion}")
print(f"Audio Types: {audio_types}")
return {
"total_duration": total_duration,
"total_gap_duration": total_gap_duration,
"emotion_changes": emotion_counts,
"most_frequent_emotion": most_frequent_emotion,
"audio_types": audio_types,
"languages":languages
}
# 分析音频数据
class AnalyzeAudioNone:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"json":(any_type,),},
}
RETURN_TYPES = (any_type,)
RETURN_NAMES = ("result",)
FUNCTION = "run"
CATEGORY = "♾️Mixlab/Audio"
def run(self,json):
result=analyze_audio_data(json)
return (result,)
class SpeechRecognition:
@classmethod
def INPUT_TYPES(s):
+278 -5
View File
@@ -1,4 +1,6 @@
import openai
from swarm import Swarm, Agent
import time
import urllib.error
import re,json,os,string,random
@@ -786,9 +788,12 @@ class JsonRepair:
def INPUT_TYPES(s):
return {
"required": {
"json_string":("STRING", {"forceInput": True,}),
"key":("STRING", {"multiline": False,"dynamicPrompts": False,"default": ""}),
}
"json_string":("STRING", {"forceInput": True,}),
"key":("STRING", {"multiline": False,"dynamicPrompts": False,"default": ""}),
},
"optional":{
"json_string2":("STRING", {"forceInput": True,})
},
}
INPUT_IS_LIST = False
@@ -800,8 +805,11 @@ class JsonRepair:
CATEGORY = "♾️Mixlab/GPT"
def run(self, json_string,key=""):
def run(self, json_string,key="",json_string2=None):
if not isinstance(json_string, str):
json_string=json.dumps(json_string)
json_string=extract_json_strings(json_string)
# print(json_string)
good_json_string = repair_json(json_string)
@@ -809,6 +817,20 @@ class JsonRepair:
# 将 JSON 字符串解析为 Python 对象
data = json.loads(good_json_string)
if json_string2!=None:
if not isinstance(json_string2, str):
json_string2=json.dumps(json_string2)
json_string2=extract_json_strings(json_string2)
# print(json_string)
good_json_string2 = repair_json(json_string2)
# 将 JSON 字符串解析为 Python 对象
data2 = json.loads(good_json_string2)
data={**data, **data2}
v=""
if key!="" and (key in data):
v=data[key]
@@ -816,4 +838,255 @@ class JsonRepair:
# 将 Python 对象转换回 JSON 字符串,确保中文字符不被转义
json_str_with_chinese = json.dumps(data, ensure_ascii=False)
return (json_str_with_chinese,v,)
return (json_str_with_chinese,v,)
# 以下为固定提示词的LLM节点示例
class SimulateDevDesignDiscussions:
def __init__(self):
# self.__client = OpenAI()
self.session_history = [] # 用于存储会话历史的列表
# self.seed=0
self.system_content="You are ChatGPT, a large language model trained by OpenAI. Answer as concisely as possible."
@classmethod
def INPUT_TYPES(cls):
model_list=[
"gpt-4o",
"gpt-4o-2024-05-13",
"gpt-4",
"gpt-4-0314",
"gpt-4-0613",
"qwen-turbo",
"qwen-plus",
"qwen-long",
"qwen-max",
"qwen-max-longcontext",
"glm-4",
"glm-3-turbo",
"moonshot-v1-8k",
"moonshot-v1-32k",
"moonshot-v1-128k",
"deepseek-chat",
"Qwen/Qwen2-7B-Instruct",
"THUDM/glm-4-9b-chat",
"01-ai/Yi-1.5-9B-Chat-16K"
]
return {
"required": {
"subject": ("STRING", {"multiline": True,"dynamicPrompts": False}),
"model": ( model_list,
{"default": model_list[0]}),
"api_url":(list(llm_apis_dict.keys()),
{"default": list(llm_apis_dict.keys())[0]}),
},
"optional":{
"api_key":("STRING", {"forceInput": True,}),
"custom_model_name":("STRING", {"forceInput": True,}), #适合自定义model
"custom_api_url":("STRING", {"forceInput": True,}), #适合自定义model
},
}
RETURN_TYPES = ("STRING",)
RETURN_NAMES = ("text",)
FUNCTION = "generate_contextual_text"
CATEGORY = "♾️Mixlab/GPT"
INPUT_IS_LIST = False
OUTPUT_IS_LIST = (False,)
def generate_contextual_text(self,
subject,
model,
api_url,
api_key=None,
custom_model_name=None,
custom_api_url=None,
):
# 设置黄色文本的ANSI转义序列
YELLOW = "\033[33m"
# 重置文本颜色的ANSI转义序列
RESET = "\033[0m"
if custom_model_name!=None:
model=custom_model_name
api_url=llm_apis_dict[api_url] if api_url in llm_apis_dict else ""
if custom_api_url!=None:
api_url=custom_api_url
if api_key==None:
api_key="lm_studio"
print("api_key,api_url",api_key,api_url)
#
if is_azure_url(api_url):
client=azure_client(api_key,api_url)
else:
# 根据用户选择的模型,设置相应的接口和模型名称
if model == "glm-4" :
client = ZhipuAI_client(api_key) # 使用 Zhipuai 的接口
print('using Zhipuai interface')
else :
client = openai_client(api_key,api_url) # 使用 ChatGPT 的接口
# 以下为多智能体框架
client = Swarm(client=client)
# 定义两个代理:软件系统架构师和设计师
software_architect_agent = Agent(
name="Software Architect",
instructions='''用脱口秀的风格回答编程问题,简短且口语化。
输出格式
====
* 答案格式:`程序员:xxxxxxxxx`
示例
==
**输入:**
如何优化代码性能?
**输出:**
程序员:兄弟,先把那些循环里的debug信息删掉,CPU都快哭了。'''
)
designer_agent = Agent(
name="Designer",
instructions='''回答问题时,请扮演一位具有多年空间设计和用户体验设计经验的设计师。你的回答应当天马行空,但又富有深度,带有苏格拉底的思考方式,并且使用脱口秀的风格。回答要简短且非常口语化。格式如下:
设计师:\[回答内容\]
Output Format
=============
* 回答应当使用“设计师:\[回答内容\]”的格式。
* 回答应当简短、口语化,富有创意和深度。
Examples
========
**Example 1:**
主持人:你觉得未来的家会是什么样子?
设计师:未来的家?想象一下,房子会像变形金刚一样,随时变形满足你的需求。今天是健身房,明天是电影院,后天是游戏场。家不再是四面墙,而是一个随心所欲的魔法空间。
**Example 2:**
主持人:你怎么看待极简主义设计?
设计师:极简主义?就像吃寿司,去掉所有不必要的装饰,只留下最精华的部分。让空间呼吸,让心灵自由。
**Example 3:**
主持人:你觉得色彩在设计中有多重要?
设计师:色彩?哦,那可是设计的灵魂!就像人生中的调味料,一点红色让你激情澎湃,一点蓝色让你心如止水。色彩决定了空间的情绪基调。'''
)
# 定义一个函数,用于转移问题到designer_agent
def transfer_to_designer_agent():
return designer_agent
# 将转移函数添加到软件系统架构师和设计师的函数列表中
software_architect_agent.functions.append(transfer_to_designer_agent)
# 问题生成
host_agent = Agent(
name="Host",
instructions='''
为播客的主持人生成4到5个问题,这些问题有些是针对设计师问的,有些是针对程序员问的。
* 主持人:你知道如何开发一款APP产品,从想法到上线吗?
* 主持人:站在设计师的角度,你怎么看?
* 主持人:不知道程序员又是怎么想的呢?
* 主持人:感谢大家的参与,今天收获蛮大的
Steps
=====
1. 确定问题的对象:设计师或程序员。
2. 根据对象设计相关的问题,确保问题的多样性和深度。
3. 整理问题,使其符合播客主持人的风格和语气。
Output Format
=============
问题列表,每个问题以“主持人:”开头,不要出现序号。
Examples
========
* 主持人:作为一名设计师,你是如何开始一个新项目的?
* 主持人:程序员在开发过程中遇到的最大挑战是什么?
* 主持人:设计师在团队协作中扮演什么角色?
* 主持人:程序员如何确保代码的质量和稳定性?
* 主持人:感谢大家的参与,今天的讨论非常有意义。
Notes
=====
* 确保问题针对不同的角色(设计师和程序员)。
* 保持问题的多样性,涵盖从项目开始到完成的各个阶段。
* 确保问题能引导出深入的讨论和见解。
''')
response = client.run(agent=host_agent, messages=[{
"role":"user",
"content":f"主题是‘{subject}’"
}],model_override=model)
content=response.messages[-1]["content"]
print(f"{YELLOW}{content}{RESET}")
texts=content.split("\n")
# texts='''
# 主持人:你知道如何开发一款APP产品,从想法到上线吗?
# 主持人:站在设计师的角度,你怎么看?
# 主持人:不知道程序员又是怎么想的呢?
# 主持人:感谢大家的参与,今天收获蛮大的
# '''.split("\n")
messages=[]
texts = [text.strip() for text in texts if text.strip()]
result=[]
for text in texts:
messages.append({
"role": "user",
"content": text
})
# 运行客户端,使用软件系统架构师作为初始代理
response = client.run(agent=software_architect_agent, messages=messages,model_override=model)
print(f"{text}")
result.append(text)
# 输出最后一个响应消息的内容
content=response.messages[-1]["content"]
print(f"{YELLOW}{content}{RESET}")
result.append(content)
messages.append({
"role":"assistant",
"content":content
})
return ("\n".join(result),)
+32 -2
View File
@@ -1,4 +1,5 @@
# Referenced some code:https://github.com/IuvenisSapiens/ComfyUI_MiniCPM-V-2_6-int4
# https://github.com/CY-CHENYUE/ComfyUI-MiniCPM-Plus
import os
import torch
@@ -35,6 +36,7 @@ class MiniCPM_VQA_Simple:
"images": ("IMAGE",),
"text": ("STRING", {"default": "", "multiline": True}),
"seed": ("INT", {"default": -1}), # add seed parameter, default is -1
"extract_keywords":("BOOLEAN", {"default": False}),
"temperature": (
"FLOAT",
{
@@ -46,7 +48,9 @@ class MiniCPM_VQA_Simple:
}
RETURN_TYPES = ("STRING",)
RETURN_TYPES = ("STRING","STRING",)
RETURN_NAMES = ("result","keywords",)
FUNCTION = "inference"
CATEGORY = "♾️Mixlab/Image"
@@ -55,6 +59,7 @@ class MiniCPM_VQA_Simple:
images,
text,
seed, # add seed parameter, default is -1
extract_keywords,
temperature,
keep_model_loaded,
):
@@ -90,6 +95,7 @@ class MiniCPM_VQA_Simple:
torch_dtype=torch.bfloat16 if self.bf16_support else torch.float16,
)
with torch.no_grad():
images = images.permute([0, 3, 1, 2])
images = [ToPILImage()(img).convert("RGB") for img in images]
@@ -113,6 +119,30 @@ class MiniCPM_VQA_Simple:
# max_new_tokens=max_new_tokens,
**params,
)
keyword_result=""
if extract_keywords:#extract_keywords
keyword_prompt = f"""Please extract keywords from the following text, including all occurrences of language (e.g. Chinese, English, etc.):
[[[{result}]]]
Please list the keywords extracted, separated by commas. Make sure to include all important words, no matter what language. For English words, please keep the original case."""
keyword_msgs = [{'role': 'user', 'content': keyword_prompt}]
keyword_result = self.model.chat(
image=None,
msgs=keyword_msgs,
tokenizer=self.tokenizer,
sampling=True,
# top_k=top_k,
# top_p=top_p,
temperature=temperature,
# repetition_penalty=repetition_penalty,
# max_new_tokens=max_new_tokens,
**params,
)
print("keyword_result",keyword_result)
# offload model to GPU
# self.model = self.model.to(torch.device("cpu"))
# self.model.eval()
@@ -124,4 +154,4 @@ class MiniCPM_VQA_Simple:
torch.cuda.empty_cache() # release GPU memory
torch.cuda.ipc_collect()
# print(result)
return (result,)
return (result,keyword_result,)
+27 -16
View File
@@ -48,6 +48,9 @@ def format_to_srt(channel_id, start_time_ms, end_time_ms, asr_result):
pattern = r"<\|(.+?)\|><\|(.+?)\|><\|(.+?)\|><\|(.+?)\|>(.+)"
match = re.match(pattern,asr_result)
print('#format_to_srt',match,asr_result)
if match==None:
return None, None, None, None,None,start_time,end_time,None
lang, emotion, audio_type, itn, text = match.groups()
# 😊 表示高兴,😡 表示愤怒,😔 表示悲伤。对于音频事件,🎼 表示音乐,😀 表示笑声,👏 表示掌声
@@ -115,16 +118,17 @@ class SenseVoiceProcessor:
part[1],
asr_result)
results.append({
"language":lang,
"emotion":emotion,
"audio_type":audio_type,
"itn":itn,
"srt_content":srt_content,
"start_time":start_time,
"end_time":end_time,
"text":text
})
if lang!=None:
results.append({
"language":lang,
"emotion":emotion,
"audio_type":audio_type,
"itn":itn,
"srt_content":srt_content,
"start_time":start_time,
"end_time":end_time,
"text":text
})
self.vad.vad.all_reset_detection()
pbar.update(1) # 更新进度条
@@ -168,8 +172,9 @@ class SenseVoiceNode:
OUTPUT_NODE = True
FUNCTION = "run"
RETURN_TYPES = (any_type,)
RETURN_NAMES = ("result",)
RETURN_TYPES = (any_type,"STRING","STRING","FLOAT",)
RETURN_NAMES = ("result","srt","text","total_seconds",)
def run(self,audio,device,language,num_threads,use_int8,use_itn ):
@@ -200,16 +205,22 @@ class SenseVoiceNode:
if 'waveform' in audio and 'sample_rate' in audio:
waveform = audio['waveform']
sample_rate = audio['sample_rate']
# print("Original shape:", waveform.shape) # 打印原始形状
if waveform.ndim == 3 and waveform.shape[0] == 1: # 检查是否为三维且 batch_size 为 1
waveform = waveform.squeeze(0) # 移除 batch_size 维度
waveform_numpy = waveform.numpy().transpose(1, 0) # 转换为 (num_samples, num_channels)
else:
raise ValueError("Unexpected waveform dimensions")
_sample_rate = audio['sample_rate']
print("waveform.shape:", waveform.shape)
total_length_seconds = waveform.shape[1] / sample_rate
results=self.processor.process_audio(waveform_numpy, _sample_rate, language, use_itn)
waveform_numpy = waveform.numpy().transpose(1, 0) # 转换为 (num_samples, num_channels)
results=self.processor.process_audio(waveform_numpy, sample_rate, language, use_itn)
return (results,)
srt_content="\n".join([s['srt_content'] for s in results])
text="\n".join([s['text'] for s in results])
return (results,srt_content,text,total_length_seconds,)
+173
View File
@@ -0,0 +1,173 @@
import os,re
import sys,time
from pathlib import Path
import torchaudio
import hashlib
import torch
import folder_paths
import comfy.utils
from faster_whisper import WhisperModel
class AnyType(str):
"""A special class that is always equal in not equal comparisons. Credit to pythongosssss"""
def __ne__(self, __value: object) -> bool:
return False
any_type = AnyType("*")
def get_model_dir(m):
try:
return folder_paths.get_folder_paths(m)[0]
except:
return os.path.join(folder_paths.models_dir, m)
whisper_model_path=get_model_dir('whisper')
model_sizes=[
d for d in os.listdir(whisper_model_path) if os.path.isdir(
os.path.join(whisper_model_path, d)
) and os.path.isfile(os.path.join(os.path.join(whisper_model_path, d), "config.json"))
]
class LoadWhisperModel:
def __init__(self):
self.model = None
self.device="cuda" if torch.cuda.is_available() else "cpu"
self.model_size=model_sizes[0]
self.compute_type='float16'
@classmethod
def INPUT_TYPES(s):
return {"required": {
"model_size": (model_sizes,),
"device": (["auto","cpu"],),
"compute_type": (["float16","int8_float16","int8"],),
},
}
RETURN_TYPES = ("WHISPER",)
RETURN_NAMES = ("whisper_model",)
FUNCTION = "run"
CATEGORY = "♾️Mixlab/Audio/Whisper"
INPUT_IS_LIST = False
OUTPUT_IS_LIST = (False,)
def run(self,model_size,device,compute_type):
if device=="auto" and self.device!='cuda':
self.device="cuda" if torch.cuda.is_available() else "cpu"
self.model=None
if device=='cpu' and self.device!='cpu':
self.device="cpu"
self.model=None
if model_size!= self.model_size:
self.model_size=model_size
self.model=None
if compute_type!=self.compute_type:
self.compute_type=compute_type
self.model=None
if self.model==None:
self.model = WhisperModel(
os.path.join(whisper_model_path, self.model_size),
device=self.device,
compute_type=self.compute_type
)
return (self.model,)
class WhisperTranscribe:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"whisper_model": ("WHISPER",),
"audio": ("AUDIO",),
},
}
RETURN_TYPES = (any_type,"STRING","STRING","FLOAT",)
RETURN_NAMES = ("result","srt","text","total_seconds",)
FUNCTION = "run"
CATEGORY = "♾️Mixlab/Audio/Whisper"
INPUT_IS_LIST = False
# OUTPUT_IS_LIST = (False,False,False,)
def run(self,whisper_model,audio):
if 'audio_path' in audio and (not 'waveform' in audio):
waveform, sample_rate = torchaudio.load(audio['audio_path'])
waveform=waveform.mean(0)
total_length_seconds = waveform.shape[0] / sample_rate
waveform=waveform.numpy()
elif 'waveform' in audio and 'sample_rate' in audio:
print("Original shape:", audio["waveform"].shape, isinstance(audio["waveform"], torch.Tensor)) # 打印原始形状
waveform = audio["waveform"].squeeze(0) # Remove the added batch dimension
sample_rate = audio["sample_rate"]
# if audio_sf != sampling_rate:
# waveform = torchaudio.functional.resample(
# waveform, orig_freq=audio_sf, new_freq=sampling_rate
# )
waveform=waveform.mean(0)
total_length_seconds = waveform.shape[0] / sample_rate
waveform=waveform.numpy() #whisper_model.transcribe 旧版不支持直接传tensor,先用numpy
segments, info = whisper_model.transcribe(waveform, beam_size=5)
print("Detected language '%s' with probability %f" % (info.language, info.language_probability))
# Function to format time for SRT
def format_time(seconds):
millis = int((seconds - int(seconds)) * 1000)
hours, remainder = divmod(int(seconds), 3600)
minutes, seconds = divmod(remainder, 60)
return f"{hours:02}:{minutes:02}:{seconds:02},{millis:03}"
# Prepare SRT content as a string
results = []
for i, segment in enumerate(segments):
start_time = format_time(segment.start)
end_time = format_time(segment.end)
srt_content = f"{i + 1}\n"
srt_content += f"{start_time} --> {end_time}\n"
text=segment.text.strip()
srt_content += f"{text}\n\n"
start_time=segment.start
end_time=segment.end
results.append({
"srt_content":srt_content,
"start_time":start_time,
"end_time":end_time,
"text":text,
"language":[info.language]
})
srt_content="\n".join([s['srt_content'] for s in results])
text="\n".join([s['text'] for s in results])
return (results,srt_content,text,total_length_seconds,)
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "comfyui-mixlab-nodes"
description = "3D, ScreenShareNode & FloatingVideoNode, SpeechRecognition & SpeechSynthesis, GPT, LoadImagesFromLocal, Layers, Other Nodes, ..."
version = "0.43.0"
version = "0.45.0"
license = "MIT"
dependencies = ["numpy", "pyOpenSSL", "watchdog", "opencv-python-headless", "matplotlib", "openai", "simple-lama-inpainting", "clip-interrogator==0.6.0", "transformers>=4.36.0", "lark-parser", "imageio-ffmpeg", "rembg[gpu]", "omegaconf==2.3.0", "Pillow>=9.5.0", "einops==0.7.0", "trimesh>=4.0.5", "huggingface-hub", "scikit-image"]
+8 -1
View File
@@ -27,4 +27,11 @@ scenedetect[opencv-headless]
hydra-core>=1.3.2
loralib>=0.1.2
natsort>=8.4.0
# simple-lama-inpainting
#simple-lama-inpainting
git+https://github.com/shadowcz007/SenseVoice-python.git
faster_whisper
git+https://github.com/openai/swarm.git
+1 -1
View File
@@ -3,7 +3,7 @@ import { app } from '../../../scripts/app.js'
const repoOwner = 'shadowcz007' // 替换为仓库的所有者
const repoName = 'comfyui-mixlab-nodes' // 替换为仓库的名称
const version = 'v0.43.0'
const version = 'v0.45.0'
fetch(`https://api.github.com/repos/${repoOwner}/${repoName}/releases/latest`)
.then(response => response.json())
+1 -1
View File
@@ -544,7 +544,7 @@ async function getCustomnodeMappings () {
const data = (await get_nodes_map()).data
window._nodes_maps = data
}
console.log('#getCustomnodeMappings', window._nodes_maps)
// console.log('#getCustomnodeMappings', window._nodes_maps)
for (let url in window._nodes_maps) {
let n = window._nodes_maps[url]
for (let node of n[0]) {
@@ -0,0 +1,416 @@
{
"last_node_id": 9,
"last_link_id": 8,
"nodes": [
{
"id": 3,
"type": "TextInput_",
"pos": [
137,
420
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "STRING",
"type": "STRING",
"links": [
2
],
"shape": 3,
"slot_index": 0
}
],
"title": "使用 Azure OpenAI",
"properties": {
"Node name for S&R": "TextInput_"
},
"widgets_values": [
"https://mixcopilot.openai.azure.com"
]
},
{
"id": 2,
"type": "KeyInput",
"pos": [
144,
257
],
"size": {
"0": 315,
"1": 70
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "key",
"type": "STRING",
"links": [
1
],
"shape": 3,
"slot_index": 0
}
],
"title": "使用你自己的key",
"properties": {
"Node name for S&R": "KeyInput"
},
"widgets_values": [
null,
null
]
},
{
"id": 6,
"type": "MultiPersonPodcast",
"pos": [
1099,
480
],
"size": [
481.8963185574753,
268.61682945154007
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "speaker",
"type": "SPEAKER",
"link": 7,
"slot_index": 0
},
{
"name": "text",
"type": "STRING",
"link": 4,
"widget": {
"name": "text"
}
}
],
"outputs": [
{
"name": "audio_list",
"type": "AUDIO",
"links": null,
"shape": 3
},
{
"name": "audio",
"type": "AUDIO",
"links": [
8
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "MultiPersonPodcast"
},
"widgets_values": [
"小明:大家好,欢迎收听本周的《AI新动态》。我是主持人小明,今天我们有两位嘉宾,分别是小李和小王。大家跟听众打个招呼吧!\n小李:大家好,我是小李,很高兴今天能和大家聊聊最新的AI动态。\n小王:大家好,我是小王,也很期待今天的讨论。",
0,
0,
0,
0,
false,
1
]
},
{
"id": 7,
"type": "LoadSpeaker",
"pos": [
584,
567
],
"size": {
"0": 315,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "speaker",
"type": "SPEAKER",
"links": [
6
],
"shape": 3,
"slot_index": 0
}
],
"title": "opus",
"properties": {
"Node name for S&R": "LoadSpeaker"
},
"widgets_values": [
"opus_00001"
]
},
{
"id": 1,
"type": "SimulateDevDesignDiscussions",
"pos": [
611,
201
],
"size": [
391.9864763335838,
217.95792637114943
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "api_key",
"type": "STRING",
"link": 1,
"widget": {
"name": "api_key"
}
},
{
"name": "custom_model_name",
"type": "STRING",
"link": null,
"widget": {
"name": "custom_model_name"
}
},
{
"name": "custom_api_url",
"type": "STRING",
"link": 2,
"widget": {
"name": "custom_api_url"
}
}
],
"outputs": [
{
"name": "text",
"type": "STRING",
"links": [
3,
4
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "SimulateDevDesignDiscussions"
},
"widgets_values": [
"数字艺术好看吗?",
"gpt-4o",
"openai",
"",
"",
""
]
},
{
"id": 8,
"type": "RenameSpeaker",
"pos": [
593,
681
],
"size": {
"0": 315,
"1": 58
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "speaker",
"type": "SPEAKER",
"link": 6
}
],
"outputs": [
{
"name": "speaker",
"type": "SPEAKER",
"links": [
7
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "RenameSpeaker"
},
"widgets_values": [
"主持人"
]
},
{
"id": 5,
"type": "ShowTextForGPT",
"pos": [
1071,
128
],
"size": [
624.2005965936271,
279.47889630613906
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "text",
"type": "STRING",
"link": 3,
"widget": {
"name": "text"
}
},
{
"name": "output_dir",
"type": "STRING",
"link": null,
"widget": {
"name": "output_dir"
}
}
],
"outputs": [
{
"name": "STRING",
"type": "STRING",
"links": null,
"shape": 6
}
],
"properties": {
"Node name for S&R": "ShowTextForGPT"
},
"widgets_values": [
"",
"",
"* 主持人:作为一名设计师,你如何定义“好看”的数字艺术?\n设计师:好看的数字艺术?就像你在沙漠中看到绿洲的那一刻,它能吸引你的眼球,抓住你的心,它能传达情感,让人产生共鸣。可能是颜色的对撞,也可能是形状的魔法,总之,它让你想多看几眼,还想收藏到你的精神博物馆里。\n* 主持人:程序员,你们在开发支持数字艺术的软件时,如何确保用户体验的直观性和美观性?\n程序员:哎呀,这可是门艺术活啊!这时候我们可不像写代码那样呆板,想象力飞起来。我们会尽量让界面简洁好用,不搞那些让人摸不着头脑的功能。动效啥的也要调校好,太多就变花里胡哨了,太少用户觉得干巴巴。最重要的是,多听设计师的,他们可是颜值担当啊!\n* 主持人:站在设计师的角度,你觉得技术如何影响了数字艺术的表现力?\n设计师:技术啊,那可是我们的魔法棒!有了高端的硬件和软件,我们可以在屏幕上玩出各种花样,大到宇宙,小到细胞,想象力在技术的加持下,才能飞得更高更远。不管是3D渲染,还是AR互动,技术就是让我们的创意从草图变成现实的桥梁,让我们画布上的每一个像素都能发光。\n* 主持人:不知道程序员又是怎么看待数字艺术的后台开发和前端展示关系的呢?\n程序员:后端和前端就像魔法师和舞台演员。后端是幕后默默挥舞魔法杖,搞定数据处理啊、服务器啥的,让那台机器运转得顺溜。前端呢,就是站在舞台中央光彩夺目,把数据和功能打包成美美的界面展示给用户。说白了,后端是灵魂,前端是颜值,两个缺一不可,配合得好才是真正的艺术!\n* 主持人:感谢大家的参与,今天关于数字艺术的讨论让我受益匪浅。\n程序员:不客气,代码和艺术的碰撞总是火花四射!\n\n设计师:没错,灵感和技术结合,才能创作出让人惊艳的作品。期待下次再聊!"
]
},
{
"id": 9,
"type": "PreviewAudio",
"pos": [
1740,
453
],
"size": {
"0": 315,
"1": 76
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "AUDIO",
"link": 8
}
],
"properties": {
"Node name for S&R": "PreviewAudio"
},
"widgets_values": [
null
]
}
],
"links": [
[
1,
2,
0,
1,
0,
"STRING"
],
[
2,
3,
0,
1,
2,
"STRING"
],
[
3,
1,
0,
5,
0,
"STRING"
],
[
4,
1,
0,
6,
1,
"STRING"
],
[
6,
7,
0,
8,
0,
"SPEAKER"
],
[
7,
8,
0,
6,
0,
"SPEAKER"
],
[
8,
6,
1,
9,
0,
"AUDIO"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 1.3310000000000006,
"offset": [
-361.773434010461,
27.423855709687306
]
}
},
"version": 0.4
}