Compare commits

..
Author SHA1 Message Date
shadowcz007 edd7af986d update 2024-10-12 10:44:39 +08:00
shadowcz007 36ef7d25ef Update ui_mixlab.js 2024-10-05 12:31:22 +08:00
shadowcz007 b766b8b65d Update SenseVoice.py 2024-10-03 11:13:14 +08:00
shadowcz007 6579ff20b4 json_string2 2024-10-03 11:12:22 +08:00
shadowcz007 2fbee59c3e fixbug 2024-10-03 11:12:10 +08:00
shadowcz007 d3aaa19148 Update ChatGPT.py 2024-10-02 17:07:10 +08:00
shadowcz007 e32a3675fc Update Whisper.py 2024-10-02 17:05:43 +08:00
shadowcz007 b72e7dda08 Update Audio.py 2024-10-02 17:05:35 +08:00
shadowcz007 0f77f28a95 Update Whisper.py 2024-10-02 16:41:46 +08:00
shadowcz007 289f83675b Update SenseVoice.py 2024-10-02 16:41:44 +08:00
shadowcz007 36633b4c72 update 2024-10-02 14:09:08 +08:00
shadowcz007 4f45457811 Update extension-node-map.json 2024-10-02 11:05:33 +08:00
shadowcz007 c39890cd64 MiniCPM_VQA_Simple add extract_keywords 2024-10-02 11:05:20 +08:00
shadowcz007 90f1e49263 Merge branch 'main' of https://github.com/shadowcz007/comfyui-mixlab-nodes 2024-10-02 09:34:42 +08:00
shadowcz007 9a1cf205db Update requirements.txt 2024-10-02 09:34:18 +08:00
shadow 45eacb6a50 Merge pull request #342 from shadowcz007/SenseVoice
Update requirements.txt
2024-10-02 09:23:38 +08:00
shadow 59f654fa39 Merge pull request #341 from shadowcz007/SenseVoice
Sense voice
2024-10-01 23:17:37 +08:00
11 changed files with 1016 additions and 85 deletions
+13 -2
View File
@@ -1006,7 +1006,7 @@ from .nodes.ImageNode import DepthViewer_,ImageBatchToList_,ImageListToBatch_,Co
# from .nodes.Vae import VAELoader,VAEDecode
from .nodes.ScreenShareNode import ScreenShareNode,FloatingVideo
from .nodes.Audio import AudioPlayNode,SpeechRecognition,SpeechSynthesis
from .nodes.Audio import AudioPlayNode,SpeechRecognition,SpeechSynthesis,AnalyzeAudioNone
from .nodes.Utils import CreateJsonNode,KeyInput,IncrementingListNode,ListSplit,CreateLoraNames,CreateSampler_names,CreateCkptNames,CreateSeedNode,TESTNODE_,TESTNODE_TOKEN,AppInfo,IntNumber,FloatSlider,TextInput,ColorInput,FontInput,TextToNumber,DynamicDelayProcessor,LimitNumber,SwitchByIndex,MultiplicationNode
from .nodes.Mask import PreviewMask_,MaskListReplace,MaskListMerge,OutlineMask,FeatheredMask
@@ -1103,6 +1103,7 @@ NODE_CLASS_MAPPINGS = {
"SpeechRecognition":SpeechRecognition,
"SpeechSynthesis":SpeechSynthesis,
"AudioPlay":AudioPlayNode,
"AnalyzeAudio":AnalyzeAudioNone,
# Text
"TextToNumber":TextToNumber,
@@ -1220,6 +1221,7 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"SpeechSynthesis":"SpeechSynthesis ♾️Mixlab",
"SpeechRecognition":"SpeechRecognition ♾️Mixlab",
"AudioPlay":"Preview Audio ♾️Mixlab",
"AnalyzeAudio":"Analyze Audio ♾️Mixlab",
# Utils
"DynamicDelayProcessor":"DynamicDelayByText ♾️Mixlab",
@@ -1433,11 +1435,20 @@ try:
from .nodes.SenseVoice import SenseVoiceNode
logging.info('SenseVoice.available')
NODE_CLASS_MAPPINGS['SenseVoiceNode']=SenseVoiceNode
NODE_DISPLAY_NAME_MAPPINGS["SenseVoiceNode"]= "Sense Voice"
NODE_DISPLAY_NAME_MAPPINGS["SenseVoiceNode"]= "Sense Voice ♾️Mixlab"
except Exception as e:
logging.info('SenseVoice.available False' )
try:
from .nodes.Whisper import LoadWhisperModel,WhisperTranscribe
logging.info('Whisper.available')
NODE_CLASS_MAPPINGS['LoadWhisperModel_']=LoadWhisperModel
NODE_CLASS_MAPPINGS['WhisperTranscribe_']=WhisperTranscribe
NODE_DISPLAY_NAME_MAPPINGS["LoadWhisperModel_"]= "Load Whisper Model ♾️Mixlab"
NODE_DISPLAY_NAME_MAPPINGS["WhisperTranscribe_"]= "Whisper Transcribe ♾️Mixlab"
except Exception as e:
logging.info('Whisper.available False' )
logging.info('\033[93m -------------- \033[0m')
File diff suppressed because it is too large Load Diff
+95
View File
@@ -3,6 +3,101 @@ import os
import folder_paths
import torchaudio
class AnyType(str):
"""A special class that is always equal in not equal comparisons. Credit to pythongosssss"""
def __ne__(self, __value: object) -> bool:
return False
any_type = AnyType("*")
def analyze_audio_data(audio_data):
total_duration = 0
total_gap_duration = 0
emotion_counts = {}
audio_types = set()
languages = set()
for i, entry in enumerate(audio_data):
# Calculate the duration of each audio segment
start_time = entry['start_time']
end_time = entry['end_time']
duration = end_time - start_time
total_duration += duration
# Count the emotions
if "emotion" in entry:
emotion = entry['emotion']
if emotion in emotion_counts:
emotion_counts[emotion] += 1
else:
emotion_counts[emotion] = 1
# Collect the audio types
if "audio_type" in entry:
audio_types.add(entry['audio_type'])
if "language" in entry:
languages.add(entry['language'])
# Calculate gap duration if not the last entry
if i < len(audio_data) - 1:
next_start_time = audio_data[i + 1]['start_time']
gap_duration = next_start_time - end_time
if gap_duration > 0:
total_gap_duration += gap_duration
# Get the most frequent emotion
if len(emotion_counts.keys())>0:
most_frequent_emotion = max(emotion_counts, key=emotion_counts.get)
else:
most_frequent_emotion=None
# Convert audio_types set to list for better readability
audio_types = list(audio_types)
languages=list(languages)
# Print the results
print(f"Total Effective Duration: {total_duration:.2f} seconds")
print(f"Total Gap Duration: {total_gap_duration:.2f} seconds")
print(f"Emotion Changes: {emotion_counts}")
print(f"Most Frequent Emotion: {most_frequent_emotion}")
print(f"Audio Types: {audio_types}")
return {
"total_duration": total_duration,
"total_gap_duration": total_gap_duration,
"emotion_changes": emotion_counts,
"most_frequent_emotion": most_frequent_emotion,
"audio_types": audio_types,
"languages":languages
}
# 分析音频数据
class AnalyzeAudioNone:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"json":(any_type,),},
}
RETURN_TYPES = (any_type,)
RETURN_NAMES = ("result",)
FUNCTION = "run"
CATEGORY = "♾️Mixlab/Audio"
def run(self,json):
result=analyze_audio_data(json)
return (result,)
class SpeechRecognition:
@classmethod
def INPUT_TYPES(s):
+24 -4
View File
@@ -786,9 +786,12 @@ class JsonRepair:
def INPUT_TYPES(s):
return {
"required": {
"json_string":("STRING", {"forceInput": True,}),
"key":("STRING", {"multiline": False,"dynamicPrompts": False,"default": ""}),
}
"json_string":("STRING", {"forceInput": True,}),
"key":("STRING", {"multiline": False,"dynamicPrompts": False,"default": ""}),
},
"optional":{
"json_string2":("STRING", {"forceInput": True,})
},
}
INPUT_IS_LIST = False
@@ -800,8 +803,11 @@ class JsonRepair:
CATEGORY = "♾️Mixlab/GPT"
def run(self, json_string,key=""):
def run(self, json_string,key="",json_string2=None):
if not isinstance(json_string, str):
json_string=json.dumps(json_string)
json_string=extract_json_strings(json_string)
# print(json_string)
good_json_string = repair_json(json_string)
@@ -809,6 +815,20 @@ class JsonRepair:
# 将 JSON 字符串解析为 Python 对象
data = json.loads(good_json_string)
if json_string2!=None:
if not isinstance(json_string2, str):
json_string2=json.dumps(json_string2)
json_string2=extract_json_strings(json_string2)
# print(json_string)
good_json_string2 = repair_json(json_string2)
# 将 JSON 字符串解析为 Python 对象
data2 = json.loads(good_json_string2)
data={**data, **data2}
v=""
if key!="" and (key in data):
v=data[key]
+32 -2
View File
@@ -1,4 +1,5 @@
# Referenced some code:https://github.com/IuvenisSapiens/ComfyUI_MiniCPM-V-2_6-int4
# https://github.com/CY-CHENYUE/ComfyUI-MiniCPM-Plus
import os
import torch
@@ -35,6 +36,7 @@ class MiniCPM_VQA_Simple:
"images": ("IMAGE",),
"text": ("STRING", {"default": "", "multiline": True}),
"seed": ("INT", {"default": -1}), # add seed parameter, default is -1
"extract_keywords":("BOOLEAN", {"default": False}),
"temperature": (
"FLOAT",
{
@@ -46,7 +48,9 @@ class MiniCPM_VQA_Simple:
}
RETURN_TYPES = ("STRING",)
RETURN_TYPES = ("STRING","STRING",)
RETURN_NAMES = ("result","keywords",)
FUNCTION = "inference"
CATEGORY = "♾️Mixlab/Image"
@@ -55,6 +59,7 @@ class MiniCPM_VQA_Simple:
images,
text,
seed, # add seed parameter, default is -1
extract_keywords,
temperature,
keep_model_loaded,
):
@@ -90,6 +95,7 @@ class MiniCPM_VQA_Simple:
torch_dtype=torch.bfloat16 if self.bf16_support else torch.float16,
)
with torch.no_grad():
images = images.permute([0, 3, 1, 2])
images = [ToPILImage()(img).convert("RGB") for img in images]
@@ -113,6 +119,30 @@ class MiniCPM_VQA_Simple:
# max_new_tokens=max_new_tokens,
**params,
)
keyword_result=""
if extract_keywords:#extract_keywords
keyword_prompt = f"""Please extract keywords from the following text, including all occurrences of language (e.g. Chinese, English, etc.):
[[[{result}]]]
Please list the keywords extracted, separated by commas. Make sure to include all important words, no matter what language. For English words, please keep the original case."""
keyword_msgs = [{'role': 'user', 'content': keyword_prompt}]
keyword_result = self.model.chat(
image=None,
msgs=keyword_msgs,
tokenizer=self.tokenizer,
sampling=True,
# top_k=top_k,
# top_p=top_p,
temperature=temperature,
# repetition_penalty=repetition_penalty,
# max_new_tokens=max_new_tokens,
**params,
)
print("keyword_result",keyword_result)
# offload model to GPU
# self.model = self.model.to(torch.device("cpu"))
# self.model.eval()
@@ -124,4 +154,4 @@ class MiniCPM_VQA_Simple:
torch.cuda.empty_cache() # release GPU memory
torch.cuda.ipc_collect()
# print(result)
return (result,)
return (result,keyword_result,)
+27 -16
View File
@@ -48,6 +48,9 @@ def format_to_srt(channel_id, start_time_ms, end_time_ms, asr_result):
pattern = r"<\|(.+?)\|><\|(.+?)\|><\|(.+?)\|><\|(.+?)\|>(.+)"
match = re.match(pattern,asr_result)
print('#format_to_srt',match,asr_result)
if match==None:
return None, None, None, None,None,start_time,end_time,None
lang, emotion, audio_type, itn, text = match.groups()
# 😊 表示高兴,😡 表示愤怒,😔 表示悲伤。对于音频事件,🎼 表示音乐,😀 表示笑声,👏 表示掌声
@@ -115,16 +118,17 @@ class SenseVoiceProcessor:
part[1],
asr_result)
results.append({
"language":lang,
"emotion":emotion,
"audio_type":audio_type,
"itn":itn,
"srt_content":srt_content,
"start_time":start_time,
"end_time":end_time,
"text":text
})
if lang!=None:
results.append({
"language":lang,
"emotion":emotion,
"audio_type":audio_type,
"itn":itn,
"srt_content":srt_content,
"start_time":start_time,
"end_time":end_time,
"text":text
})
self.vad.vad.all_reset_detection()
pbar.update(1) # 更新进度条
@@ -168,8 +172,9 @@ class SenseVoiceNode:
OUTPUT_NODE = True
FUNCTION = "run"
RETURN_TYPES = (any_type,)
RETURN_NAMES = ("result",)
RETURN_TYPES = (any_type,"STRING","STRING","FLOAT",)
RETURN_NAMES = ("result","srt","text","total_seconds",)
def run(self,audio,device,language,num_threads,use_int8,use_itn ):
@@ -200,16 +205,22 @@ class SenseVoiceNode:
if 'waveform' in audio and 'sample_rate' in audio:
waveform = audio['waveform']
sample_rate = audio['sample_rate']
# print("Original shape:", waveform.shape) # 打印原始形状
if waveform.ndim == 3 and waveform.shape[0] == 1: # 检查是否为三维且 batch_size 为 1
waveform = waveform.squeeze(0) # 移除 batch_size 维度
waveform_numpy = waveform.numpy().transpose(1, 0) # 转换为 (num_samples, num_channels)
else:
raise ValueError("Unexpected waveform dimensions")
_sample_rate = audio['sample_rate']
print("waveform.shape:", waveform.shape)
total_length_seconds = waveform.shape[1] / sample_rate
results=self.processor.process_audio(waveform_numpy, _sample_rate, language, use_itn)
waveform_numpy = waveform.numpy().transpose(1, 0) # 转换为 (num_samples, num_channels)
results=self.processor.process_audio(waveform_numpy, sample_rate, language, use_itn)
return (results,)
srt_content="\n".join([s['srt_content'] for s in results])
text="\n".join([s['text'] for s in results])
return (results,srt_content,text,total_length_seconds,)
+173
View File
@@ -0,0 +1,173 @@
import os,re
import sys,time
from pathlib import Path
import torchaudio
import hashlib
import torch
import folder_paths
import comfy.utils
from faster_whisper import WhisperModel
class AnyType(str):
"""A special class that is always equal in not equal comparisons. Credit to pythongosssss"""
def __ne__(self, __value: object) -> bool:
return False
any_type = AnyType("*")
def get_model_dir(m):
try:
return folder_paths.get_folder_paths(m)[0]
except:
return os.path.join(folder_paths.models_dir, m)
whisper_model_path=get_model_dir('whisper')
model_sizes=[
d for d in os.listdir(whisper_model_path) if os.path.isdir(
os.path.join(whisper_model_path, d)
) and os.path.isfile(os.path.join(os.path.join(whisper_model_path, d), "config.json"))
]
class LoadWhisperModel:
def __init__(self):
self.model = None
self.device="cuda" if torch.cuda.is_available() else "cpu"
self.model_size=model_sizes[0]
self.compute_type='float16'
@classmethod
def INPUT_TYPES(s):
return {"required": {
"model_size": (model_sizes,),
"device": (["auto","cpu"],),
"compute_type": (["float16","int8_float16","int8"],),
},
}
RETURN_TYPES = ("WHISPER",)
RETURN_NAMES = ("whisper_model",)
FUNCTION = "run"
CATEGORY = "♾️Mixlab/Audio/Whisper"
INPUT_IS_LIST = False
OUTPUT_IS_LIST = (False,)
def run(self,model_size,device,compute_type):
if device=="auto" and self.device!='cuda':
self.device="cuda" if torch.cuda.is_available() else "cpu"
self.model=None
if device=='cpu' and self.device!='cpu':
self.device="cpu"
self.model=None
if model_size!= self.model_size:
self.model_size=model_size
self.model=None
if compute_type!=self.compute_type:
self.compute_type=compute_type
self.model=None
if self.model==None:
self.model = WhisperModel(
os.path.join(whisper_model_path, self.model_size),
device=self.device,
compute_type=self.compute_type
)
return (self.model,)
class WhisperTranscribe:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"whisper_model": ("WHISPER",),
"audio": ("AUDIO",),
},
}
RETURN_TYPES = (any_type,"STRING","STRING","FLOAT",)
RETURN_NAMES = ("result","srt","text","total_seconds",)
FUNCTION = "run"
CATEGORY = "♾️Mixlab/Audio/Whisper"
INPUT_IS_LIST = False
# OUTPUT_IS_LIST = (False,False,False,)
def run(self,whisper_model,audio):
if 'audio_path' in audio and (not 'waveform' in audio):
waveform, sample_rate = torchaudio.load(audio['audio_path'])
waveform=waveform.mean(0)
total_length_seconds = waveform.shape[0] / sample_rate
waveform=waveform.numpy()
elif 'waveform' in audio and 'sample_rate' in audio:
print("Original shape:", audio["waveform"].shape, isinstance(audio["waveform"], torch.Tensor)) # 打印原始形状
waveform = audio["waveform"].squeeze(0) # Remove the added batch dimension
sample_rate = audio["sample_rate"]
# if audio_sf != sampling_rate:
# waveform = torchaudio.functional.resample(
# waveform, orig_freq=audio_sf, new_freq=sampling_rate
# )
waveform=waveform.mean(0)
total_length_seconds = waveform.shape[0] / sample_rate
waveform=waveform.numpy() #whisper_model.transcribe 旧版不支持直接传tensor,先用numpy
segments, info = whisper_model.transcribe(waveform, beam_size=5)
print("Detected language '%s' with probability %f" % (info.language, info.language_probability))
# Function to format time for SRT
def format_time(seconds):
millis = int((seconds - int(seconds)) * 1000)
hours, remainder = divmod(int(seconds), 3600)
minutes, seconds = divmod(remainder, 60)
return f"{hours:02}:{minutes:02}:{seconds:02},{millis:03}"
# Prepare SRT content as a string
results = []
for i, segment in enumerate(segments):
start_time = format_time(segment.start)
end_time = format_time(segment.end)
srt_content = f"{i + 1}\n"
srt_content += f"{start_time} --> {end_time}\n"
text=segment.text.strip()
srt_content += f"{text}\n\n"
start_time=segment.start
end_time=segment.end
results.append({
"srt_content":srt_content,
"start_time":start_time,
"end_time":end_time,
"text":text,
"language":[info.language]
})
srt_content="\n".join([s['srt_content'] for s in results])
text="\n".join([s['text'] for s in results])
return (results,srt_content,text,total_length_seconds,)
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "comfyui-mixlab-nodes"
description = "3D, ScreenShareNode & FloatingVideoNode, SpeechRecognition & SpeechSynthesis, GPT, LoadImagesFromLocal, Layers, Other Nodes, ..."
version = "0.43.0"
version = "0.44.0"
license = "MIT"
dependencies = ["numpy", "pyOpenSSL", "watchdog", "opencv-python-headless", "matplotlib", "openai", "simple-lama-inpainting", "clip-interrogator==0.6.0", "transformers>=4.36.0", "lark-parser", "imageio-ffmpeg", "rembg[gpu]", "omegaconf==2.3.0", "Pillow>=9.5.0", "einops==0.7.0", "trimesh>=4.0.5", "huggingface-hub", "scikit-image"]
+5 -2
View File
@@ -27,6 +27,9 @@ scenedetect[opencv-headless]
hydra-core>=1.3.2
loralib>=0.1.2
natsort>=8.4.0
# simple-lama-inpainting
git+https://github.com/shadowcz007/SenseVoice-python.git
#simple-lama-inpainting
git+https://github.com/shadowcz007/SenseVoice-python.git
faster_whisper
+1 -1
View File
@@ -3,7 +3,7 @@ import { app } from '../../../scripts/app.js'
const repoOwner = 'shadowcz007' // 替换为仓库的所有者
const repoName = 'comfyui-mixlab-nodes' // 替换为仓库的名称
const version = 'v0.43.0'
const version = 'v0.44.0'
fetch(`https://api.github.com/repos/${repoOwner}/${repoName}/releases/latest`)
.then(response => response.json())
+1 -1
View File
@@ -544,7 +544,7 @@ async function getCustomnodeMappings () {
const data = (await get_nodes_map()).data
window._nodes_maps = data
}
console.log('#getCustomnodeMappings', window._nodes_maps)
// console.log('#getCustomnodeMappings', window._nodes_maps)
for (let url in window._nodes_maps) {
let n = window._nodes_maps[url]
for (let node of n[0]) {